{"as_of":"2026-08-14T13:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a4da0ec37ceb61264277be121bed0fd8a9e9b12e28dbf7ae01923fa586817583","coverage":[{"denominator":71,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":71,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:15:48.995779Z","state":"measured"},{"denominator":71,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":71,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.19535/citation-record","integrity":"/paper/2505.19535/integrity","json":"/paper/2505.19535/citation-record.json","paper":"/paper/2505.19535"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:58.842603Z","title":"Tune-a-video: One-shot tuning of image diffusion models for text-to-video generation,","venue":null,"work_id":"d971d2d6-49de-449c-baf5-a1dd7fa13567","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:43.651761Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:42cddb948942dae959918bbe205f26825ff9153e38151335816c44ba813d640d","observation_id":"49259069-4357-497b-af56-bb497360a7ed","resolution":{"observed_at":"2026-08-07T14:15:58.979116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:58.544759Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation,","venue":null,"work_id":"5c4c451b-86a9-43eb-ad04-795c1a76b7c0","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:43.742055Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:324c255e2a9ef090959ba6904c4e6e3780290ab9ac89d847dcad3a5363d9e217","observation_id":"d726bc0c-741b-47db-9faa-edd19846277f","resolution":{"observed_at":"2026-08-07T14:15:58.676070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:58.212805Z","title":"Text2video-zero: Text-to-image diffusion models are zero-shot video generators,","venue":null,"work_id":"60ec42af-9790-43e7-9f6b-91717cc95542","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:43.841821Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:f6386454cee522d6ba456a1db793ff8c598a4d6856b5c0b2c2ef3881338360b5","observation_id":"2f1b4a46-ee54-44d6-9c4c-13ed27702cc6","resolution":{"observed_at":"2026-08-07T14:15:58.429716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:58.028487Z","title":"Ccedit: Creative and controllable video editing via diffusion models,","venue":null,"work_id":"6dda3ce7-8d16-46d3-aad6-347e61f1b06a","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:43.901242Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:412fbba121ec59dc8b76c25f5a04878273964a72cddfeb2f122615df3af8ac9a","observation_id":"60d1a271-6100-4bc2-8999-e9635db4082b","resolution":{"observed_at":"2026-08-07T14:15:58.111035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:57.784564Z","title":"Controlvideo: Training-free controllable text-to-video generation,","venue":null,"work_id":"b0bf88dc-8d21-4477-8e0e-39c095d205fe","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.024176Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:036b29285a088f9e633514af879238fef6e7e680d4ec1f59d24b195d6ca91efa","observation_id":"fb830f27-c3ed-4215-ad0d-1187587f0dab","resolution":{"observed_at":"2026-08-07T14:15:57.919016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:57.627990Z","title":"Fatezero: Fusing attentions for zero-shot text-based video editing,","venue":null,"work_id":"d69b191d-cb4a-4fde-9538-da6089af33fa","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.195828Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:b9fd7e9009ef43aea25a6ef12ced903c4859ec836ec7ef46424a3dbbbdd94d35","observation_id":"b9a50780-125f-42b2-988f-e6e5d402579e","resolution":{"observed_at":"2026-08-07T14:15:57.713443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:57.457694Z","title":"Flatten: optical flow-guided attention for consistent text-to-video editing,","venue":null,"work_id":"f66f63e5-96af-4510-9f9f-235c5bd2da11","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.339750Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:c2845497d95f17ef7700a6a61cecabbcfd42269a50d3c95bc1e7d01922e4cced","observation_id":"05c9366b-5047-46ac-8824-75e3d23a5552","resolution":{"observed_at":"2026-08-07T14:15:57.555158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:57.242692Z","title":"Fresco: Spatial-temporal correspondence for zero-shot video translation,","venue":null,"work_id":"aac59014-0c7a-409a-8059-bee1c8775f98","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.443157Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:b05b39f75369af07f4ac310ab981f014cc1de5fdf5a85cb7e074a152f9199e8d","observation_id":"2941ad73-01ea-4f6a-b494-4538eea50770","resolution":{"observed_at":"2026-08-07T14:15:57.301203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:57.079280Z","title":"Pix2video: Video editing using image diffusion,","venue":null,"work_id":"9be8cf9e-f38e-4d48-8d25-2cd6dc3cf809","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.574580Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:9ab92c95146e998d78a7ec97b0774983457d7309ec00e122b91acef0cfaee65d","observation_id":"47e9861c-8a2d-4354-b017-20dbd203fab1","resolution":{"observed_at":"2026-08-07T14:15:57.154855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.878966Z","title":"Rave: Randomized noise shuffling for fast and consistent video editing with diffusion models,","venue":null,"work_id":"f40233b8-cb97-4ef0-90a4-142ee7f3f70e","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.666645Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:a6cc5be98665d0e8f7da2c4a1ef3efaeb87c8eb5d7d4a9b8585c9c1d3c6ee22f","observation_id":"77254bda-3784-410c-9557-33efcdda1cfb","resolution":{"observed_at":"2026-08-07T14:15:56.972919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.770806Z","title":"Slicedit: Zero-shot video editing with text-to-image diffusion models using spatio-temporal slices,","venue":null,"work_id":"f8dc0995-2847-42b6-a894-6ad3a52c7f27","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.788333Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:2adbc5b6a0d41dd058314165addab29bddf88a2342b3b3b52c5b136254ce66d3","observation_id":"6d9acb26-1789-4300-a8ff-33f813aa1edb","resolution":{"observed_at":"2026-08-07T14:15:56.824036Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.643642Z","title":"Zero-shot video editing using off-the-shelf image diffusion models,","venue":null,"work_id":"55e1f8b2-c913-4bb3-8b61-9097eae0499f","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.869181Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:136251eac4fec199015a45253f69a1534ecf73f53b353c77d239eab9898eef5f","observation_id":"ad892360-db24-4767-b16a-1ac2c692ca94","resolution":{"observed_at":"2026-08-07T14:15:56.720375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.473058Z","title":"Video quality assessment: A comprehensive survey,","venue":null,"work_id":"b1409ab5-6de1-4390-b588-711952994e6b","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.964072Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:44fd855d8f588c4e277ec65c2fb5cf0f76675e3f2353d0c3e4574c3166b7119b","observation_id":"76e4353c-1fb6-48f1-ab9d-c5f18c8268da","resolution":{"observed_at":"2026-08-07T14:15:56.573558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.347184Z","title":"Exploring video quality assessment on user generated contents from aesthetic and technical perspectives,","venue":null,"work_id":"be44e27f-a6d5-403a-863d-fb4878b3f519","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.062144Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:d1b01585b0f814e566a7185bebd0b18d16b5de6ce2bbaf45e614f4581fa84d84","observation_id":"e4796152-584e-4f9a-81d4-47c9342fe8d8","resolution":{"observed_at":"2026-08-07T14:15:56.401381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.222297Z","title":"Fast-vqa: Efficient end-to-end video quality assessment with fragment sampling,","venue":null,"work_id":"b7382e6a-53ad-4d76-b767-f3cdf6f4b8c8","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.140757Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:e43de9539604497e3563364b1f590f61d02d4ff61c000ccf11409a5398cfabb9","observation_id":"a5d76651-a57b-4067-ab29-cab838afc2b4","resolution":{"observed_at":"2026-08-07T14:15:56.275262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.995545Z","title":"A deep learning based no-reference quality assessment model for ugc videos,","venue":null,"work_id":"86f9226f-7870-40af-b6f8-32cee121e802","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.223546Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:21668d77c3703f5cdf1e6645a7903f88478c722b5ab5ee48c6ea88f2a86d7b65","observation_id":"141736e6-6844-4070-9a96-8377261db06e","resolution":{"observed_at":"2026-08-07T14:15:56.115372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.857344Z","title":"Quality assessment of in-the-wild videos,","venue":null,"work_id":"9d5723e2-83b9-48c6-a927-c908b8e17626","year":2019},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.303727Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:6028babf14b876b66fe42138949fd8bc435a05f5dd90fbd98ba41e09c4f05453","observation_id":"075bcaea-ff09-4a68-b72e-52c2829fc70e","resolution":{"observed_at":"2026-08-07T14:15:55.923996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.720812Z","title":"Ugc-vqa: Benchmarking blind video quality assessment for user generated content,","venue":null,"work_id":"caf3c255-3f4d-41a5-9be7-57b4ccc5998f","year":2020},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.396067Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:324a36d58149b1e66f170d01afd7e5ca9950dfc07174bf9fdb7faa5d46020cbe","observation_id":"db5585b8-a3c6-4ad3-8baa-48519105547e","resolution":{"observed_at":"2026-08-07T14:15:55.755136Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.610988Z","title":"Ve-bench: Subjective-aligned benchmark suite for text-driven video editing quality assessment,","venue":null,"work_id":"630887df-6b7a-4ed5-8147-193a140b71c3","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.493803Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:17a384b905290f1189c1cb9d19de4743b663baa5f1a3c60d848cf6b562d9ec31","observation_id":"015e49cc-725b-4faa-83a4-9c08c135c34f","resolution":{"observed_at":"2026-08-07T14:15:55.658082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.475777Z","title":"Subjective-aligned dataset and metric for text-to-video quality assessment,","venue":null,"work_id":"8e5252ed-49b6-4eaf-bf23-1d34196b4286","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.569424Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:bb98ee20c89314dbba420dfe230e1ced9220c921a5e37dec1b17e8dc88585359","observation_id":"67dc21db-bbe0-4bd6-a807-dea04770e5b2","resolution":{"observed_at":"2026-08-07T14:15:55.566921Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.326401Z","title":"Aigv-assessor: Benchmarking and evaluating the perceptual quality of text-to-video generation with lmm,","venue":null,"work_id":"43af6dbb-f8d4-4c18-896f-3b79489a519d","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.660369Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:e9bfddb45baeaf531250b76191e167ae6746d1f624768ce25ab943bcfe1706bf","observation_id":"382ed968-9186-4e37-a265-2431fb21154b","resolution":{"observed_at":"2026-08-07T14:15:55.397791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.208835Z","title":"Cvpr 2023 text guided video editing competition,","venue":null,"work_id":"d17d9636-2d58-4dd5-b4c9-6de06ec6836c","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.715156Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:0c837f7b5392d2780a167b422e9c67cde7079f5d9d26697e0c9ed014c20b9a3e","observation_id":"37e63f66-13fb-4635-8466-3d8fe8f257ed","resolution":{"observed_at":"2026-08-07T14:15:55.263392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.063178Z","title":"Harnessing the power of llms in practice: A survey on chatgpt and beyond,","venue":null,"work_id":"881d4db7-c4a1-4366-a07d-858d622b98ef","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.772617Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:5b6140de12284ba6b6d0da71e4f12ea2922976232f391dc1aaf6060d63fdbc6f","observation_id":"8567a7a2-5497-4f38-bc97-57584fa67d93","resolution":{"observed_at":"2026-08-07T14:15:55.127769Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.873579Z","title":"Señorita- 2m: A high-quality instruction-based dataset for general video editing by video specialists,","venue":null,"work_id":"3b5132b3-a237-404b-8baa-8f3c99dd343c","year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.864925Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:c8c43deb797a5e3ff59e0cab5f72f9475f2db4d29d7ddd34ec1e8d9eeedd2d46","observation_id":"909a4cb3-8658-48fb-9a50-7f876484b7a9","resolution":{"observed_at":"2026-08-07T14:15:54.957500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.766512Z","title":"The 2017 davis challenge on video object segmentation,","venue":null,"work_id":"837fb387-96e6-4e06-87c2-5a3abd6169e3","year":2017},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.952544Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:6963448344c4dc3a1a28b939febe396701b83fb45a161a9a9eaad926150bd315","observation_id":"fc602e93-00f8-4104-9789-58f37f95a2e1","resolution":{"observed_at":"2026-08-07T14:15:54.811565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.652967Z","title":"The kinetics human action video dataset,","venue":null,"work_id":"c32ab599-c39d-4637-a5f9-0147753e1643","year":2017},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.050651Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:5095b3ffcd8fad9500ec0d7a9b72648fabdd1363a5219a320a672823834c86e3","observation_id":"b3a43984-edbb-4e33-8d52-60aa4f3a2ae7","resolution":{"observed_at":"2026-08-07T14:15:54.691934Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.501381Z","title":"JimengAI","venue":null,"work_id":"f5706de7-5780-4dc3-bb59-788eafa33701","year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.117830Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:fef8b7cbf56bf781defde8c5c9e012bce7ffb36f47e56e0a4d90283d14c84194","observation_id":"a159a87b-60dd-4c58-8742-d0082cd328ed","resolution":{"observed_at":"2026-08-07T14:15:54.573138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.336329Z","title":"Methodology for the subjective assessment of the quality of television pictures,","venue":null,"work_id":"1c4e7919-f020-4d64-99a9-56f47dc81c94","year":2012},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.186786Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:56cffe346e3a8bfd1a44210d9fedba1c68db58b9a4cbad0e07d560aaca5dc02a","observation_id":"e6e4e421-f9e6-476f-9d65-a954de5c28f3","resolution":{"observed_at":"2026-08-07T14:15:54.428466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.169105Z","title":"Blind image quality assessment based on high order statistics aggregation,","venue":null,"work_id":"a59ca9c3-6219-42b0-9a5a-770fc80eae89","year":2016},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.255534Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:f2f97af738d38791ff6a2fc1e88a0a8eb8451abdfa4c70743531dd9dbc65d175","observation_id":"c9cdea49-6a70-47ab-9afd-3a6c8b0baa73","resolution":{"observed_at":"2026-08-07T14:15:54.235971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.095797Z","title":"Learning without human scores for blind image quality assessment,","venue":null,"work_id":"511f3c93-bc3f-45ba-b1ec-3a71b00d67a0","year":2013},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.336487Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:a0bc053766d89c26b496c97e16db2fb45cf7a4a11a5396d700392980698e05fd","observation_id":"0fb5fcef-ca42-4cee-ad5b-f95dd1083574","resolution":{"observed_at":"2026-08-07T14:15:54.125353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.025445Z","title":"Making a “completely blind","venue":null,"work_id":"ebc60996-781d-427c-b8e8-b827ccc4657d","year":2013},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.415983Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:336473b966877ff5b5748904e2005bceac0516e644e932b0f099fe6b17035102","observation_id":"58668ad9-9f78-4c1b-b000-54e70d6d5511","resolution":{"observed_at":"2026-08-07T14:15:54.055030Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.983229Z","title":"Clipscore: A reference-free evaluation metric for image captioning,","venue":null,"work_id":"a96f5067-945e-4795-9b8c-cf7ec1aba230","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.508413Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:5110666a4fb1c8e5f0ace34138cdebdddf6d5d5accf6169c5be97d0c58ddba5c","observation_id":"147b5037-7061-4135-a851-e193d80a0556","resolution":{"observed_at":"2026-08-07T14:15:53.999272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.01291","last_updated":"2024-06-18T07:09:55Z","snapshot_observed_at":"2026-08-13T00:39:54.992015Z","submitted_at":"2024-04-01T17:58:06Z","title":"Evaluating Text-to-Visual Generation with Image-to-Text Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.01291","snapshot_observed_at":"2026-08-07T14:15:46.589608Z","title":"Evaluating text-to-visual generation with image-to-text generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.589608Z"},"links":{"cited_paper":"/paper/2404.01291","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:cd63f582f73fa860d033aefc6f28357fcc29f6cdc980400772a8325187f19e02","observation_id":"0082d370-eec2-4389-a745-f24588fd8b08","resolution":{"observed_at":"2026-08-07T14:15:46.589608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.08358","last_updated":"2025-04-11T08:46:49Z","snapshot_observed_at":"2026-08-12T08:40:53.220571Z","submitted_at":"2025-04-11T08:46:49Z","title":"LMM4LMM: Benchmarking and Evaluating Large-multimodal Image Generation with LMMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.08358","snapshot_observed_at":"2026-08-07T14:15:46.647524Z","title":"Lmm4lmm: Benchmarking and evaluating large-multimodal image generation with lmms,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.647524Z"},"links":{"cited_paper":"/paper/2504.08358","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:7051631e0b6bd33184f02921eee119a6d9ebce46c617dd8e7837df89c488cc11","observation_id":"fa6aed1c-e178-4085-a73d-c7cc251bcf1c","resolution":{"observed_at":"2026-08-07T14:15:46.647524Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:46.707838Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.707838Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:87f3537eecc4cebe39b00947086b3f2592822f14ed071371a3a8924cb86932f2","observation_id":"9720f9e0-4f87-4be1-ad52-edb8df975867","resolution":{"observed_at":"2026-08-07T14:15:46.707838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.877089Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale,","venue":null,"work_id":"a47b4189-53f8-4abd-9e23-56d2b755f39b","year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.796154Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:dc2e423ab94217bf2b51d2e44cb0f61185451280e153bf22d3713f0957dc23f0","observation_id":"b6ab865e-ded1-44f6-b9ce-1167a8e66290","resolution":{"observed_at":"2026-08-07T14:15:53.957123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.748793Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale,","venue":null,"work_id":"3b0db710-266e-4b83-9009-7f53a2f250d1","year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.889880Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:a6a312c6e10d99b206a200e75c5d14122a23a7983a7c217c7209b5809b169e57","observation_id":"514cfcbf-2390-4329-9e73-c59ea492ffa2","resolution":{"observed_at":"2026-08-07T14:15:53.808845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.01601","last_updated":"2021-06-11T09:36:50Z","snapshot_observed_at":"2026-08-06T23:46:59.006674Z","submitted_at":"2021-05-04T16:17:21Z","title":"MLP-Mixer: An all-MLP Architecture for Vision","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.01601","snapshot_observed_at":"2026-08-07T14:15:46.972275Z","title":"Mlp-mixer: An all-mlp architecture for vision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.972275Z"},"links":{"cited_paper":"/paper/2105.01601","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:511b2043f4e5cca5636392aee8f0b09dad346819091d16fde68c316c8631d91d","observation_id":"2a2ac725-4e2f-4ca5-9db3-93e2c7a3cef6","resolution":{"observed_at":"2026-08-07T14:15:46.972275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.10270","last_updated":"2022-06-23T10:51:04Z","snapshot_observed_at":"2026-08-13T19:00:44.169969Z","submitted_at":"2021-06-18T17:58:20Z","title":"How to train your ViT? Data, Augmentation, and Regularization in Vision Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.10270","snapshot_observed_at":"2026-08-07T14:15:47.047930Z","title":"How to train your vit? data, augmentation, and regularization in vision transformers,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.047930Z"},"links":{"cited_paper":"/paper/2106.10270","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:817f653fc41d98e099bd9283c1b4ea0569ee44386b999e05b4a39469129df618","observation_id":"9fc44e78-5762-44f8-b03f-022e60c60f95","resolution":{"observed_at":"2026-08-07T14:15:47.047930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.01548","last_updated":"2022-03-13T18:58:43Z","snapshot_observed_at":"2026-08-13T19:10:52.020157Z","submitted_at":"2021-06-03T02:08:03Z","title":"When Vision Transformers Outperform ResNets without Pre-training or Strong Data Augmentations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.01548","snapshot_observed_at":"2026-08-07T14:15:47.098749Z","title":"When vision transformers outperform resnets without pretraining or strong data augmentations,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.098749Z"},"links":{"cited_paper":"/paper/2106.01548","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:bb3c5c1405d229c55d8a45a08a717081839857d740b4fcaf04a9f8dba4187a08","observation_id":"8febaff3-2ab2-4aa5-b1d0-617cdeb1442c","resolution":{"observed_at":"2026-08-07T14:15:47.098749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.500783Z","title":"Surrogate gap minimization improves sharpness-aware training,","venue":null,"work_id":"216e36e9-7542-4ffd-afd3-f87aa1dcbad7","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.146349Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:11bb84e1d0554c856cc22d63ce6518821d26f3e74cb6afd3a389ccc54c188709","observation_id":"0db7e7a8-44cb-495a-88e1-b7ee9eb8c641","resolution":{"observed_at":"2026-08-07T14:15:53.597132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.341023Z","title":"Lit: Zero-shot transfer with locked-image text tuning,","venue":null,"work_id":"62cd42ed-3e31-4080-8103-5cae877e2b67","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.196903Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:e57f60ba90b0cd1ab74209eae239471758dd55cd858eaac19f0deed1c921793e","observation_id":"761438d2-410b-48c9-a6b6-59cb3f6a55f7","resolution":{"observed_at":"2026-08-07T14:15:53.438080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-07T14:15:47.249382Z","title":"Qwen-vl: A versa- tile vision-language model for understanding, localization, text reading, and beyond,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.249382Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:c4e3ae451350f323babf1e343b71f798a1e841814972fe4939c60a072e6cad14","observation_id":"1664f839-5ebe-4a7c-bb8f-b579c62ce79f","resolution":{"observed_at":"2026-08-07T14:15:47.249382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.125365Z","title":"Mlp-net: Multilayer perceptron fusion network for infrared small target detection,","venue":null,"work_id":"b479b6b3-fba3-4ea4-b019-93e78834e81f","year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.303914Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:d8a35616be5c235fb8baac0b3e434ecafc7bc6ff0521f4bef2fe11f1dd82ecab","observation_id":"8adb38d0-4237-4e12-b5bd-e0eb6b3fe445","resolution":{"observed_at":"2026-08-07T14:15:53.213820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:52.877682Z","title":"Lora: Low-rank adaptation of large language models,","venue":null,"work_id":"400d4d09-f4f4-43c2-bd09-2ee5f7acb93a","year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.364459Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:374dd6391da10488011756071863b8e9948a4a5a2b67a9272ec19d4708116fca","observation_id":"65a60f14-dd82-401e-a048-46d7db18602a","resolution":{"observed_at":"2026-08-07T14:15:52.987257Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:52.667618Z","title":"Imagereward: learning and evaluating human preferences for text-to-image generation,","venue":null,"work_id":"fc2eaa4f-8117-4358-96a0-cc338a019a4d","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.419599Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:b3dbb5b92469a05c823090107afeb7cce0f6a27166175c124e746bce1043d47e","observation_id":"d53a406b-4189-488a-99ed-9272ada4c374","resolution":{"observed_at":"2026-08-07T14:15:52.781780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:52.498420Z","title":"Blip: Bootstrapping language-image pre-training for unified vision- language understanding and generation,","venue":null,"work_id":"648bb444-4dbe-437d-957a-6bed02372e90","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.497508Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:cef74e61e6a049dd3d3eb176a9ce1f29631b681c82e29545b5e7eeca09e47539","observation_id":"c95245d3-7d6c-402a-aa39-aec719e0a493","resolution":{"observed_at":"2026-08-07T14:15:52.561473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:52.275626Z","title":"Pick-a-pic: An open dataset of user preferences for text-to-image generation,","venue":null,"work_id":"793a3e92-43f8-45fb-bd11-465c62940c8b","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.556046Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:12842ab2349ee02547b31a5f7fd1bb5958270159b7fc963934e038ec2988ca09","observation_id":"4f49277c-9c6d-4030-a140-a3ee6b1f4dde","resolution":{"observed_at":"2026-08-07T14:15:52.397957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:52.078567Z","title":"Building cnn-based models for image aesthetic score prediction using an ensemble,","venue":null,"work_id":"8602fdd4-ddd7-400c-a98a-c8bf11d5679f","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.618742Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:108e508ea4f7c1fd49c76f8a976d6a8d6bb9e89f770bf651b01766e4ec136e22","observation_id":"11517637-b9b8-4fb5-a164-d3be77ecd053","resolution":{"observed_at":"2026-08-07T14:15:52.174546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:51.827935Z","title":"Llava-next: A strong zero-shot video understanding model,","venue":null,"work_id":"a15bfb9d-fd24-4cef-9f26-bb4e30feecf9","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.671411Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:3f3e645bd780c9a87f24d65665cba783bb191705c05f534ec3bc490ee5311583","observation_id":"f8b2128c-3ff8-4a5c-b537-99a95b3bab61","resolution":{"observed_at":"2026-08-07T14:15:51.951951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:51.615212Z","title":"Internvideo: General video foundation models via generative and discriminative learning,","venue":null,"work_id":"3f99eea7-0861-4aca-ae0b-0b37619ca54a","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.723918Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:d00f7b702d5efe2489c8bc6647a16febde7ac957ad2267b210a85e14017ed82e","observation_id":"92317989-0820-4ead-b432-384fdfa782e0","resolution":{"observed_at":"2026-08-07T14:15:51.692478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-08-13T11:53:19.464291Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13106","snapshot_observed_at":"2026-08-07T14:15:47.793673Z","title":"Videollama 3: Frontier multimodal foundation models for image and video understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.793673Z"},"links":{"cited_paper":"/paper/2501.13106","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:bce924b7775489143dc7be966d1ac34374a0000c1e6cf60294dc2d576dacb886","observation_id":"b97954b8-5d3e-4053-a390-0a9036c01158","resolution":{"observed_at":"2026-08-07T14:15:47.793673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:51.406576Z","title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks,","venue":null,"work_id":"5a1b81ee-3750-4dbf-bd8d-5287c1d8b4ec","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.842241Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:003232a4d7186233886b2cc32ebefa9d06ed75e42346c8269182cae9a0713fc7","observation_id":"16b6cae0-883d-4759-b36e-d73e8eec2a68","resolution":{"observed_at":"2026-08-07T14:15:51.511413Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:51.273234Z","title":"mplug-owl3: Towards long image-sequence understanding in multi-modal large language models,","venue":null,"work_id":"79e9afc2-d23f-4768-ad7e-565fc9f86238","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.905970Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:72993c9ce27ee6f70ebe5611edd8e8466920fbeeb17c070d87cea8c4ecd36618","observation_id":"dbc1d83f-1c7a-43b1-995f-c66bac60f979","resolution":{"observed_at":"2026-08-07T14:15:51.344286Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:47.959442Z","title":"No-reference image quality assessment in the spatial domain,","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.959442Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:b94dcdaf15034e0d5580178bc19c16bfa2cc717cdb20cf3f80044639139fbb8b","observation_id":"caaec614-1d36-46d3-b353-25615a657f3e","resolution":{"observed_at":"2026-08-07T14:15:47.959442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:51.115921Z","title":"Blind image quality estimation via distortion aggravation,","venue":null,"work_id":"84b941e4-0776-4dae-b623-c918f975d868","year":2018},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.029727Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:95d6cf1130f0052febb7546a61e4aa86314a707638a7847787309dbda1823d50","observation_id":"e78e9d04-ab79-454a-9fff-8fdf3e641069","resolution":{"observed_at":"2026-08-07T14:15:51.198918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:50.910319Z","title":"Blind quality assessment based on pseudo- reference image,","venue":null,"work_id":"048bc7c2-c120-4c44-a0c7-c7611f900958","year":2018},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.090491Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:d6b39bb2165e547f09fad384b426a388007b56912a0513f3513b5224a222c6ad","observation_id":"79cbeca4-a14e-441e-be03-e04a34a93f5b","resolution":{"observed_at":"2026-08-07T14:15:51.004024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:48.138760Z","title":"Visual instruction tuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.138760Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:09e1c1256d236196d6377b689e8891e3925784808b95ccff1f9ded7a6b4ff3e7","observation_id":"403b8b0a-6ae7-426a-a442-ceabac5f51bc","resolution":{"observed_at":"2026-08-07T14:15:48.138760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07895","last_updated":"2024-07-28T19:58:08Z","snapshot_observed_at":"2026-08-13T00:09:23.835117Z","submitted_at":"2024-07-10T17:59:43Z","title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.07895","snapshot_observed_at":"2026-08-07T14:15:48.190049Z","title":"Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.190049Z"},"links":{"cited_paper":"/paper/2407.07895","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:b0f3bb4858807b75fd666263a98bace78fa915811233005c72c377ab16dae007","observation_id":"1f8005a3-abc0-497d-a663-8d9b958afd0b","resolution":{"observed_at":"2026-08-07T14:15:48.190049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:48.245414Z","title":"Llava-next: Improved reasoning, ocr, and world knowledge,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.245414Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:ae6068f6c2d5af7099b64efce3d388364290772590fb8be74d2d3270ad8d3c7e","observation_id":"f883cf17-f2ed-408c-9b71-becbad8dd081","resolution":{"observed_at":"2026-08-07T14:15:48.245414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:48.349429Z","title":"Improved baselines with visual instruction tuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.349429Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:58902ebe5cdfc583340065bf609b44e4865ee3b55d066cb2856d842a4d978239","observation_id":"a9df28ab-111c-4a68-821e-a7aaf4172c19","resolution":{"observed_at":"2026-08-07T14:15:48.349429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:50.657726Z","title":"Llava-next: Stronger llms supercharge multimodal capabilities in the wild,","venue":null,"work_id":"7551d759-9346-40ae-a132-1ba091cc6efc","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.412809Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:2d4c62dca7dab1e17a344226ca9b14d30cdd1f9a356f9d8a8537d39506557183","observation_id":"865709a4-8de1-4b48-be07-011908404bcd","resolution":{"observed_at":"2026-08-07T14:15:50.758104Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:50.430836Z","title":"Llava-next: What else influences visual instruction tuning beyond data?,","venue":null,"work_id":"d7300167-9967-4430-981c-bccc320c9abe","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.473759Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:4470599613d4684a4154eff052d4d638d89d3715bf5d2b58793b6c129f5edd95","observation_id":"8b7fa63b-bfa7-40b9-993b-d750b86428ca","resolution":{"observed_at":"2026-08-07T14:15:50.534207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-08-13T15:50:38.254753Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-07T14:15:48.537796Z","title":"Video-llama: An instruction-tuned audio-visual language model for video understanding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.537796Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:2bb56129ccf7debb198eeae2598d9c16330abb79776cc495aae981b9e316ba35","observation_id":"18370840-77bd-42b7-9442-253997febc9a","resolution":{"observed_at":"2026-08-07T14:15:48.537796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-07T14:15:48.595108Z","title":"Videollama 2: Advancing spatial-temporal modeling and audio understanding in video-llms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.595108Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:55cb5e5ad91c4fffd84ab9aa6e0e53b14ad0181385d844b6826f979c5fc938dc","observation_id":"2a484005-937b-47b3-912e-9e815524a744","resolution":{"observed_at":"2026-08-07T14:15:48.595108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:50.237089Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":"8ca92e43-7b6e-4736-9670-4a187b907905","year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.673221Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:29c59a9d4a94f348fe28e232dad4d49bc519335388df2c00520926071f035fdc","observation_id":"3e219ea3-4a27-42f1-9288-8bbe7e7f46c3","resolution":{"observed_at":"2026-08-07T14:15:50.332408Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:49.999171Z","title":"Unmasked teacher: Towards training- efficient video foundation models,","venue":null,"work_id":"b98310b4-809b-4388-acc6-9c02f8583113","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.732449Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:dc9f0f03402a6209347f12d2381ec263f3558969a08c738bf0707578696f6c4f","observation_id":"dba5ab64-a687-4600-bbb6-142a5c5a19b2","resolution":{"observed_at":"2026-08-07T14:15:50.106302Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:49.782748Z","title":"Stablevqa: A deep no-reference quality assessment model for video stability,","venue":null,"work_id":"30a4e065-0867-4c8c-94e7-2e9403e74d06","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.810810Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:47b36fa39470de9ade7e9b6f3f5ab21bd91d7a2cb183b4ce7b4f532ddf7e1f16","observation_id":"7e7a514f-68bf-41ce-aea6-7b24568ecc26","resolution":{"observed_at":"2026-08-07T14:15:49.870172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17090","last_updated":"2023-12-28T16:10:25Z","snapshot_observed_at":"2026-08-02T07:14:02.308302Z","submitted_at":"2023-12-28T16:10:25Z","title":"Q-Align: Teaching LMMs for Visual Scoring via Discrete Text-Defined Levels","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17090","snapshot_observed_at":"2026-08-07T14:15:48.881635Z","title":"Q-align: Teaching lmms for visual scoring via discrete text-defined levels,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.881635Z"},"links":{"cited_paper":"/paper/2312.17090","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:890877e70d34d85c228715529e3468285ac8fb737140882fd06684f8dc5faf1e","observation_id":"f8d41227-00e3-4ca1-8fdf-781525ee55a7","resolution":{"observed_at":"2026-08-07T14:15:48.881635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:49.541531Z","title":"Excellent","venue":null,"work_id":"b4414455-bba7-4130-b398-891fa4564aca","year":null},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.932160Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:51b2286e43c4d9b2e0e889a7b8a9ebeeba7f973338b5c010b9318fc934e8659b","observation_id":"b589e717-2e61-441e-bf33-a1400fd02fa0","resolution":{"observed_at":"2026-08-07T14:15:49.655959Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:49.309914Z","title":"Excellent","venue":null,"work_id":"a2ef180a-6a07-480f-98c7-8cef90c1a51a","year":2016},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.995779Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:902793c9736e5fa447dffde3421be0c327dec8c3f6f26ca6127ec0f26d0e81b9","observation_id":"4d17d73d-ae3a-467f-bceb-a0c55182ad0f","resolution":{"observed_at":"2026-08-07T14:15:49.409764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T09:36:17.755126Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs"},"reference_resolution":{"displayed":71,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":55},"total_outbound_references":71},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 71 of 71 outbound references and 0 inbound Pith citation observations for arXiv:2505.19535."}