{"as_of":"2026-08-23T13:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b3811e275ba66db0d06245144356ac7f01bdd462ec7ea97c71aa7ccf68168faa","coverage":[{"denominator":34,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:28:40.527544Z","state":"measured"},{"denominator":34,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":34,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.04587/citation-record","integrity":"/paper/2608.04587/integrity","json":"/paper/2608.04587/citation-record.json","paper":"/paper/2608.04587"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:48.998659Z","title":"Claude code","venue":null,"work_id":"0c4559b3-5eec-451e-a702-54dec88eec7f","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:36.114059Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:8d5781a4056f59c98607d347849a716daaafa9ab323d8e3a2c966daef1e95aa4","observation_id":"41643fe5-1ec1-4e37-a05c-32f8e3a2ce4e","resolution":{"observed_at":"2026-08-06T21:28:49.142112Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:48.694776Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding, 2024","venue":null,"work_id":"36eb975f-cfdc-4b16-9191-1cf1bbc07b6f","year":2024},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:36.190927Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:20cfb37e47c4b3484ba17883e15ab32e448491c36ef4b204682b3e864d80bd27","observation_id":"295ecae5-ed8b-44af-8672-500206bc3780","resolution":{"observed_at":"2026-08-06T21:28:48.842564Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:48.432409Z","title":"GraphVideoAgent: Enhancing long-form video understanding with entity relation graphs","venue":null,"work_id":"035755aa-af6c-4eca-8170-47b8cd52cd2c","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:36.366711Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:6b7b3a4e50bc2839508e0ac3b91fbf6407d75b5076f324e63b6a4b65d5fb76e9","observation_id":"4f29ecf3-a921-40a8-ac42-1bb7623f27c4","resolution":{"observed_at":"2026-08-06T21:28:48.535373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:48.113410Z","title":"TC- Pad’e: Trajectory-consistent Pad’e approximation for diffusion acceleration, 2026","venue":null,"work_id":"bf46a54c-c573-465c-81da-f8975e5b4996","year":2026},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:36.509438Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:8a47a63c5bbd696daad8d1027b967e7d4f3dd675b36c8397421270307b25c847","observation_id":"e2559b31-d232-4135-9f5b-fc7e8f4c6fab","resolution":{"observed_at":"2026-08-06T21:28:48.296765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:47.839115Z","title":"Diffusion probe: Generated image result prediction using CNN probes, 2026","venue":null,"work_id":"f49c2af0-a64b-43c5-a4c9-d7cbe223ec4b","year":2026},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:36.658141Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:1fb9f863c6c16fcc69aa2fb4ee0e4893b172bc5c15c1617d509589017fe69a86","observation_id":"0db988a4-d994-4f68-bc09-3ed797a2a308","resolution":{"observed_at":"2026-08-06T21:28:47.953498Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:47.558338Z","title":"simpleposter: a simple base- line for product poster generation, 2026","venue":null,"work_id":"610cbe8d-72b4-4e6f-826b-a5428c78ecf2","year":2026},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:36.783049Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:4fcbf903922c20666011950e686ac37f868ab1bd8ac5745d3ddc19377a7d5d9b","observation_id":"6fe23ce1-b231-4d07-abbf-5c41b701f574","resolution":{"observed_at":"2026-08-06T21:28:47.714457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:47.296642Z","title":"Video-mme: The first-ever comprehensive evalu- ation benchmark of multi-modal llms in video analysis,","venue":null,"work_id":"1d16c22d-45a5-4e7e-95a7-c5025cebc231","year":null},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:36.929383Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:ce88808e9e706a89deac11c83f460dc301ad8006ef5d3aeeda01d2517a211372","observation_id":"8f228e6e-fa31-42e6-bbd5-5e6520c0623f","resolution":{"observed_at":"2026-08-06T21:28:47.402941Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:46.985332Z","title":"Automated design of agentic systems","venue":null,"work_id":"5f6ced41-c754-4d69-8501-ee2402821dca","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:37.049295Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:c5774d0cc118845e64158ae0682ba7e99ec06e8378e57a15e50d37c4fc0630fe","observation_id":"470f7be9-83f6-4a80-b26c-022769ea8866","resolution":{"observed_at":"2026-08-06T21:28:47.117642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:46.733318Z","title":"RAMS-Trans: Recurrent attention multi-scale transformer for fine- grained image recognition","venue":null,"work_id":"744eb2e3-589e-4b08-ba8c-af632a05de42","year":2021},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:37.173589Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:125e2d7ed18a6c1199e0a9ac4c594813068c0cf28810afa0d1ca8e53b4854c34","observation_id":"626697cb-1653-4617-9de4-0d7476e26cb4","resolution":{"observed_at":"2026-08-06T21:28:46.859573Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:46.468447Z","title":"Perceive- to-reason: Decoupling perception and reasoning for fine-grained visual reasoning, 2026","venue":null,"work_id":"94eeda3f-11df-468c-9768-27c642a48128","year":2026},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:37.366276Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:600b9d17161720e3660d4d4b5ba577b6430da45a30117658802a4582348b63cc","observation_id":"cc9734c7-8ac5-42cc-a1f6-9edf4c1ab772","resolution":{"observed_at":"2026-08-06T21:28:46.621522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:46.243986Z","title":"Lenswalk: Agentic video understanding by planning how you see in videos,","venue":null,"work_id":"babfd3f3-346d-46ed-91ae-874f80e4a783","year":null},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:37.519822Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:0a95a2f52c7d8f2afeb216e53f680cc28596a2bfdb367f317895a214916616de","observation_id":"24315fb7-dd62-4b44-9a9c-a8dea9e0faaf","resolution":{"observed_at":"2026-08-06T21:28:46.344911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:45.976499Z","title":"Videoseek: Long-horizon video agent with tool-guided seeking, 2026","venue":null,"work_id":"a87cd19f-d059-4f2f-bca9-dd382526e580","year":2026},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:37.611766Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:1685146d19a61732831ddaa4896f6abf15d77fbd898f37c53a4b9149ea7cc1ab","observation_id":"a5cdda10-7e85-42ee-b12b-e36e956f2cfe","resolution":{"observed_at":"2026-08-06T21:28:46.078779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:45.709722Z","title":null,"venue":null,"work_id":"464517c5-53b3-4b52-af2e-e766d3212d78","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:37.728874Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:ae431853e4b3e8f9b608dedcca7d01e68e05a35c43848a774749fef19666f1d4","observation_id":"d0ffa3d8-7619-46e2-92d4-5fa776602baa","resolution":{"observed_at":"2026-08-06T21:28:45.837805Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:45.463395Z","title":"VideoMind: A chain-of-LoRA agent for temporal-grounded video reasoning","venue":null,"work_id":"1a00783c-2ddb-493a-9698-67fed47ec602","year":null},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:37.840677Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:da6720316ed2cd97eb7235f07a10f5afe1e0cd33f6ed03c7c60ca38702d5eac6","observation_id":"05b1ffe5-928f-4f5e-9da8-785ca434e1ca","resolution":{"observed_at":"2026-08-06T21:28:45.576169Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:45.216929Z","title":"Seeing, listen- ing, remembering, and reasoning: A multimodal agent with long-term memory","venue":null,"work_id":"52fb74e8-cef8-4af6-8ecf-f45bbe377090","year":2026},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:37.945615Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:8986cdaec5f7e7cb0c4c8e804ddc21ee1cc4648cc0c88c719a3e4042e58002bf","observation_id":"47fb6933-2e63-4e77-9aa8-a2ffd20ff1fb","resolution":{"observed_at":"2026-08-06T21:28:45.343957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:45.006500Z","title":"Openai codex","venue":null,"work_id":"12171b85-4e9a-45b0-a3a4-50ae3f1ace93","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:38.082147Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:8612435562681f078e36883e20c1a60b2b9846ad28e3899faf9b0001e5a93bad","observation_id":"2bb09e34-8aaa-47c7-9bcf-18cba6f85243","resolution":{"observed_at":"2026-08-06T21:28:45.098262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:44.814103Z","title":"Yuvion VL: A multimodal foundation model for adversarial content and AI safety, 2026","venue":null,"work_id":"053dd036-b035-4ca8-add3-d5c4aeccfb70","year":2026},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:38.225025Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:2b823d32f7c1339662529c02b2c8e933773268ec0ac80b881eb76646f4a835a8","observation_id":"c731efaf-d051-4e86-a67c-622478843faf","resolution":{"observed_at":"2026-08-06T21:28:44.891333Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:44.588899Z","title":"Attentive eraser: Unleashing diffusion model’s object removal potential via self-attention redirection guidance, 2025","venue":null,"work_id":"01bf0196-f735-4d00-8850-7c156f0b4ed3","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:38.334858Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:ad55517ddb513ccddd565228cf3e50db4c903b6627b517e19c0c683d5ba06982","observation_id":"e54f05ca-f5ff-4ad1-a3b1-fb7a607cba77","resolution":{"observed_at":"2026-08-06T21:28:44.690832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:44.408438Z","title":"Open- vocabulary object detection with an open corpus","venue":null,"work_id":"76a91e26-ac56-4949-9f3c-78f0300e28d4","year":2023},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:38.498419Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:bc3c4129e553e8a7b917acac4e99caaf6305bfbc24c06668535c44d260f6d639","observation_id":"97d7f9e7-081d-43f2-b00b-31d6e496ceaa","resolution":{"observed_at":"2026-08-06T21:28:44.503280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:44.190856Z","title":"Videoagent: Long-form video understand- ing with large language model as agent, 2024","venue":null,"work_id":"235e7298-a7bd-47c9-b8ce-fde8b36ecdb4","year":2024},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:38.647413Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:95c070848c3c3f60c2a899bb3c391dd60f50319ca0f49d364299712a4ee83b2a","observation_id":"f32688bb-e9ac-4add-94ee-30ae749a32ec","resolution":{"observed_at":"2026-08-06T21:28:44.301950Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:43.988584Z","title":"VideoTree: Adaptive tree-based video repre- sentation for LLM reasoning on long videos","venue":null,"work_id":"36d9abc5-ff75-4912-8b29-b7b48566199d","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:38.779462Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:8f202d5aac3c296fef91e94a19c02b45c89ba8b3f4af5411bee90f23f0fadeca","observation_id":"096b2fbf-0619-43a3-8e2e-5eb776462f7d","resolution":{"observed_at":"2026-08-06T21:28:44.076853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:43.803029Z","title":"Longvideobench: A benchmark for long-context inter- leaved video-language understanding, 2024","venue":null,"work_id":"78e8f557-a141-4982-b1c0-82e704c6d92c","year":2024},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:38.905576Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:ebe7ca45f5c23defe99cba26d25d8580e64f656728030fab74c546a316e2dcf8","observation_id":"3b62c6d3-9f7f-469d-8965-709de4b9248f","resolution":{"observed_at":"2026-08-06T21:28:43.886113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:43.566482Z","title":"Seeing but not thinking: Routing distraction in multimodal mixture- of-experts","venue":null,"work_id":"5c3a4a95-814a-4895-90ec-1bb0cfa4f28f","year":2026},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:39.051759Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:c3859fa7cdc6f13d1a36dc356a4d723e16c9d643738aae26988c3f04e2a42682","observation_id":"5a4a108e-5c1e-49cc-9ad8-0b4b5c3b020b","resolution":{"observed_at":"2026-08-06T21:28:43.662507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:43.348249Z","title":"Symphony: A cognitively-inspired multi-agent system for long-video understanding","venue":null,"work_id":"54200d2e-bb4d-47ce-8dcb-bb89e6f04bc4","year":2026},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:39.169014Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:497ea28098d81c1fcb481419f911f050f1a68a0c43d6d31703312113d28203bd","observation_id":"0a80c223-8136-4b4e-a038-0f7b5ddc348d","resolution":{"observed_at":"2026-08-06T21:28:43.479191Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:43.127951Z","title":"Worldmm: Dynamic multimodal memory agent for long video reasoning, 2025","venue":null,"work_id":"49e1f910-6dcd-4531-8689-c4633ea87161","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:39.318229Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:7070783dccaae65c05eea69e6091b7d8a590e79c54bc119f3b257738abc5eed4","observation_id":"b82e5979-21f9-45db-ac6b-316f4496fa41","resolution":{"observed_at":"2026-08-06T21:28:43.250821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:42.873579Z","title":"Hierarchical long video understanding with audiovisual entity cohesion and agentic search, 2026","venue":null,"work_id":"d28c018b-77a8-4b42-81a7-b6e21c53c4bd","year":2026},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:39.477574Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:9ee133fab47ee6189098db88883a5a47b57face30f698cd08027ba68fcf0aa7c","observation_id":"075b0a05-67bf-434d-850d-a3f0fa98ce1e","resolution":{"observed_at":"2026-08-06T21:28:42.992854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:42.577980Z","title":"Videoarm: Agentic rea- soning over hierarchical memory for long-form video understanding, 2025","venue":null,"work_id":"adb66e7a-c70b-4bfd-be51-b5d3b91a79ee","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:39.651654Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:19f45eadabfe2fceff4933b21ce36bdb4efcd504b266d695568365d1f2b4c631","observation_id":"7f201548-0994-4196-b0db-b1fd030131b9","resolution":{"observed_at":"2026-08-06T21:28:42.729710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:42.312220Z","title":"EvoAgent: Towards au- tomatic multi-agent generation via evolutionary algo- rithms","venue":null,"work_id":"6473dc2f-44ed-4066-93d2-b7f09ec12efa","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:39.773510Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:779f3fea25c4eb613267afb7b28e7ff89d40eb67f45f3233c6855a610ee9d2ea","observation_id":"a6844819-2ef1-4ee8-bdc4-a3768443b5c1","resolution":{"observed_at":"2026-08-06T21:28:42.423932Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:42.007213Z","title":"AFlow: Automating agentic workflow generation","venue":null,"work_id":"16df5ad3-3140-4374-8ead-2cba8273f5ab","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:39.889785Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:8ddb7ed7eea730579528c735027e6a34a4c5f809dfa00be9fd1c306e602d2a57","observation_id":"f8accfb5-0636-4ba8-88ef-56198da83f32","resolution":{"observed_at":"2026-08-06T21:28:42.143696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:41.740094Z","title":"OmAgent: A multi-modal agent framework for complex video understanding with task divide-and-conquer","venue":null,"work_id":"0f1204f7-bd40-4064-ad25-1198116d066a","year":2024},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:40.012914Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:45d2167b2c2476a9af8c05174203ad5a20183e00e191b29941ceb8efc6efcb2b","observation_id":"51e81c20-d3cd-4541-a8ae-40783b6e811e","resolution":{"observed_at":"2026-08-06T21:28:41.868710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:41.514461Z","title":"Deep video discov- ery: Agentic search with tool use for long-form video understanding","venue":null,"work_id":"10fcd2b0-fe6b-41a8-8bfe-a242af2c17cf","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:40.130193Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:d1a90d046202e05bd4aadfbeafbc178cd037469e79e983b0be5c5ea052cd1dcd","observation_id":"bfd64926-ce66-4c2e-8d47-79e447d5f5ec","resolution":{"observed_at":"2026-08-06T21:28:41.603928Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:41.195076Z","title":"Deep video discov- ery: Agentic search with tool use for long-form video understanding, 2025","venue":null,"work_id":"a9b945b8-f8d3-4521-b3b3-9dac2126fa20","year":2025},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:40.269165Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:a1647c197a3c0da62a2d6e22fc56caea7cc426f62b1d45a00470ec1e42f9ddf6","observation_id":"f669e676-66b0-40d6-b4b7-89d87a7f0ab8","resolution":{"observed_at":"2026-08-06T21:28:41.370951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:40.916263Z","title":"Mlvu: Benchmarking multi-task long video understanding,","venue":null,"work_id":"3d2874ca-e6cc-4461-9753-35477d3f8676","year":null},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:40.368522Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:aee14378daa977d0a620bd5d9be9fcff881ed8d88092cb79efcbba7324417b3c","observation_id":"1924b7ce-049d-4840-a9ca-b02633f8f3dc","resolution":{"observed_at":"2026-08-06T21:28:41.059931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:28:40.675385Z","title":"In fact, the earlier shape was this kind of whisk","venue":null,"work_id":"b693a59b-f03c-4ba0-91de-2f5b7731160d","year":null},"citing_paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T21:28:40.527544Z"},"links":{"citing_paper":"/paper/2608.04587"},"observation_digest":"sha256:3bf712e90b339f52ac5a8e0672f242f965e5f76bf29239108782bf7569325758","observation_id":"20973239-7458-4e6b-975b-86282cfae730","resolution":{"observed_at":"2026-08-06T21:28:40.758096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.04587","last_updated":"2026-08-05T08:50:23Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-17T18:25:46.949017Z","submitted_at":"2026-08-05T08:50:23Z","title":"MetaVideoAgent: Automated Video-Agent Evolution for Long-Form Video Understanding"},"reference_resolution":{"displayed":34,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":1,"verified_exact":0,"verified_fuzzy":33},"total_outbound_references":34},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 34 of 34 outbound references and 0 inbound Pith citation observations for arXiv:2608.04587."}