{"as_of":"2026-08-09T18:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:77d0d8751c39ca94441e52ba1a888fa0bbefe11414a97f985233920aa26a691d","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":35,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:02:29.351108Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T06:39:37.518508Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":120,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:a93cb2996d56d27d7d3baa556a6d45f2c1aa5c4f32e743f85f4fa517277e2522","observation_id":"0ea28090-2a9b-430f-a743-afc02aa2e0ea","resolution":{"observed_at":"2026-05-12T20:58:59.244814Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2410.05363","last_updated":"2024-10-07T17:56:04Z","snapshot_observed_at":"2026-08-08T09:48:34.424815Z","submitted_at":"2024-10-07T17:56:04Z","title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-18T14:39:59.870039Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2410.05363"},"observation_digest":"sha256:34f8086de520055c04cf526a2dab715fce9813725cf7839ab970857bc0fe3fc0","observation_id":"6482212f-2fea-4ee8-af46-15f76d04a960","resolution":{"observed_at":"2026-05-18T14:40:00.050951Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2410.17434","last_updated":"2024-10-22T21:21:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-22T21:21:37Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T13:53:33.585035Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2410.17434"},"observation_digest":"sha256:b2e94fa7efcd203977a57a5878172698096cdfbed3b8514c8a56a4a4cf361a01","observation_id":"67648cd9-4a98-42e9-9df2-3427496945b5","resolution":{"observed_at":"2026-05-16T13:53:33.707536Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":257,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:12a9d60a6a767590477e4524dffde1ef0697344cb0fb77644b4040dbf2aa0582","observation_id":"be600ca6-544c-4333-806e-450eb4add321","resolution":{"observed_at":"2026-05-10T13:23:58.143903Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2412.15689","last_updated":"2026-05-06T21:36:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-20T09:07:36Z","title":"DOLLAR: Few-Step Video Generation via Distillation and Latent Reward Optimization","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-23T06:57:50.897865Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2412.15689"},"observation_digest":"sha256:baa92486cebea8ec21e8849987128d5cec2f93f50cd1ca020eb6b2983d8491ce","observation_id":"9b3a2c62-41ad-4638-849f-bff192cadec6","resolution":{"observed_at":"2026-05-23T07:02:41.738013Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2501.02955","last_updated":"2026-05-12T15:02:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-06T11:57:38Z","title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-23T05:44:31.546843Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2501.02955"},"observation_digest":"sha256:da026a2d6a6c4375e4416a9e6fe6603f39ce497829df603e22c0cd9e69829591","observation_id":"d4ddbc8d-88bb-4485-b83d-33783b4f2547","resolution":{"observed_at":"2026-05-23T05:45:28.358964Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2501.05067","last_updated":"2026-04-20T07:42:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-09T08:43:57Z","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-23T06:01:00.775721Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2501.05067"},"observation_digest":"sha256:dbf5dd592ccaa4a1e78eb1eaace524b105794d02bec639c1b0b0602e95234ffe","observation_id":"e76c4e79-5a2b-4e07-83b2-a35350833ea4","resolution":{"observed_at":"2026-05-23T06:02:37.554003Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:861bf495b7b1719cdf47f3cf4607d9324f93b000bee34ab22d9008637cd4da87","observation_id":"fe182bdf-a1eb-452c-ab36-07d2a8315a47","resolution":{"observed_at":"2026-05-11T01:19:59.742495Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T15:02:29.351108Z","title":"Internvideo2: Scaling video foundation models for multimodal video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16594","last_updated":"2025-05-22T12:28:50Z","snapshot_observed_at":"2026-08-07T14:55:54.546955Z","submitted_at":"2025-05-22T12:28:50Z","title":"Temporal Object Captioning for Street Scene Videos from LiDAR Tracks","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:02:29.351108Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2505.16594"},"observation_digest":"sha256:872c9052b31d36d03802e8428566be57e2f8e74dc9a54f3027de042ff5f60e5c","observation_id":"b85f535a-1e35-4405-ac04-20df1c5043b4","resolution":{"observed_at":"2026-08-07T15:02:29.351108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T14:24:10.374002Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19125","last_updated":"2025-05-25T12:44:12Z","snapshot_observed_at":"2026-08-07T14:17:46.430576Z","submitted_at":"2025-05-25T12:44:12Z","title":"RTime-QA: A Benchmark for Atomic Temporal Event Understanding in Large Multi-modal Models","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:10.374002Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2505.19125"},"observation_digest":"sha256:4d002236dffff103671c41638b403d6b4f334c51884eaf5e7188d8d2a5608aae","observation_id":"fbc88a1a-335d-41cc-9183-f0aab1822dcd","resolution":{"observed_at":"2026-08-07T14:24:10.374002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T13:52:47.260706Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20644","last_updated":"2025-05-27T02:45:14Z","snapshot_observed_at":"2026-08-07T13:47:38.739847Z","submitted_at":"2025-05-27T02:45:14Z","title":"HCQA-1.5 @ Ego4D EgoSchema Challenge 2025","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T13:52:47.260706Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2505.20644"},"observation_digest":"sha256:4ab28944f055b866fef4de41cbb46b4be4e63f926a8ad13740d0392572c1b8c6","observation_id":"4f82f8a4-75f2-4b0b-bd9f-12cdf4e124f5","resolution":{"observed_at":"2026-08-07T13:52:47.260706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T13:48:15.401308Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding.arXiv preprint arXiv:2403.15377, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20920","last_updated":"2025-05-27T09:10:59Z","snapshot_observed_at":"2026-08-09T07:43:10.959148Z","submitted_at":"2025-05-27T09:10:59Z","title":"HuMoCon: Concept Discovery for Human Motion Understanding","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T13:48:15.401308Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2505.20920"},"observation_digest":"sha256:1eba20786e25f235929360cbe2be8e71e257946953041a0a48fdffebda94468a","observation_id":"c51851bd-3b04-4f8f-9f48-14366c20f7cd","resolution":{"observed_at":"2026-08-07T13:48:15.401308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T10:27:05.499105Z","title":"Internvideo2: Scaling video foundation models for multimodal video understanding.arXiv preprint arXiv:2403.15377, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05328","last_updated":"2025-07-22T07:00:35Z","snapshot_observed_at":"2026-08-08T19:39:48.959316Z","submitted_at":"2025-06-05T17:58:33Z","title":"AV-Reasoner: Improving and Benchmarking Clue-Grounded Audio-Visual Counting for MLLMs","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:27:05.499105Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.05328"},"observation_digest":"sha256:dae2f5853b0f5f31eb84fb78f4a18d7476b184ecd9f30533131986e8772b8231","observation_id":"5c096c6d-ac30-465c-a225-b79619168202","resolution":{"observed_at":"2026-08-07T10:27:05.499105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T10:27:30.926631Z","title":"Internvideo2: Scaling video foundation models for multimodal video understanding.arXiv preprint arXiv:2403.15377, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05336","last_updated":"2025-07-05T11:38:26Z","snapshot_observed_at":"2026-08-08T13:58:02.206064Z","submitted_at":"2025-06-05T17:59:29Z","title":"VideoMolmo: Spatio-Temporal Grounding Meets Pointing","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T10:27:30.926631Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.05336"},"observation_digest":"sha256:ab6178b7fb6ee8369f5f3990381769403ec1bfcd61969ddd6b2fd45e27553ec0","observation_id":"34c1518b-6a96-4e58-ad3a-81998148f13c","resolution":{"observed_at":"2026-08-07T10:27:30.926631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T04:07:59.061572Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11659","last_updated":"2025-06-13T10:40:23Z","snapshot_observed_at":"2026-08-09T08:19:07.203224Z","submitted_at":"2025-06-13T10:40:23Z","title":"An Empirical study on LLM-based Log Retrieval for Software Engineering Metadata Management","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T04:07:59.061572Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.11659"},"observation_digest":"sha256:51fa17b9de80051b6c0e6035f5f87816a8335cd88310690068a4cb1b290d8a5d","observation_id":"a206936a-a89b-4446-b0ca-5da22eb7bbec","resolution":{"observed_at":"2026-08-07T04:07:59.061572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T00:51:56.851307Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12585","last_updated":"2025-06-14T17:39:03Z","snapshot_observed_at":"2026-08-07T00:43:16.069936Z","submitted_at":"2025-06-14T17:39:03Z","title":"DejaVid: Encoder-Agnostic Learned Temporal Matching for Video Classification","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T00:51:56.851307Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.12585"},"observation_digest":"sha256:67e30f9c6fd56ebce5b2072fb08d1c4b1e8075db34e8b18b14e4f4e4c05c1d98","observation_id":"30471e61-9c6b-4718-a4bb-197fbf6e0fd0","resolution":{"observed_at":"2026-08-07T00:51:56.851307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T00:26:00.161920Z","title":"InternVideo2: Scaling foundation models for multimodal video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14356","last_updated":"2025-06-17T09:51:51Z","snapshot_observed_at":"2026-08-09T16:46:35.710854Z","submitted_at":"2025-06-17T09:51:51Z","title":"EVA02-AT: Egocentric Video-Language Understanding with Spatial-Temporal Rotary Positional Embeddings and Symmetric Optimization","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T00:26:00.161920Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.14356"},"observation_digest":"sha256:27a4eb624927ad9eb81886055294ffdd4cbdaa07d22d46501e46ba9bd9ae8d9c","observation_id":"0c6c2b82-1304-4e98-8301-7975bacc3446","resolution":{"observed_at":"2026-08-07T00:26:00.161920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-06T23:42:05.964749Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16450","last_updated":"2025-06-19T16:35:49Z","snapshot_observed_at":"2026-08-06T23:36:03.391214Z","submitted_at":"2025-06-19T16:35:49Z","title":"How Far Can Off-the-Shelf Multimodal Large Language Models Go in Online Episodic Memory Question Answering?","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T23:42:05.964749Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.16450"},"observation_digest":"sha256:a65556fbbecb0ab62234072fbb9496e3041890b5566eeccae140d67067a639cb","observation_id":"26244bf5-0094-4a2f-a175-3efb65d83ac2","resolution":{"observed_at":"2026-08-06T23:42:05.964749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-06T22:24:33.036457Z","title":"Internvideo2: Scaling video foundation models for multimodal video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21862","last_updated":"2025-06-27T02:29:58Z","snapshot_observed_at":"2026-08-09T10:29:21.701927Z","submitted_at":"2025-06-27T02:29:58Z","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-06T22:24:33.036457Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.21862"},"observation_digest":"sha256:b49d07cc089cfbb0c27fb6d807387b0cb80a76af1022dcde9ff87a9d4002409a","observation_id":"f2293dc2-fba5-4555-8a7b-352c33d71e45","resolution":{"observed_at":"2026-08-06T22:24:33.036457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2507.04590","last_updated":"2025-07-07T00:51:57Z","snapshot_observed_at":"2026-08-08T15:48:31.665148Z","submitted_at":"2025-07-07T00:51:57Z","title":"VLM2Vec-V2: Advancing Multimodal Embedding for Videos, Images, and Visual Documents","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-18T14:10:14.929207Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2507.04590"},"observation_digest":"sha256:81cd5f80dadf724333722bf7ea1e421719a2c426121e4c998abc372899bea464","observation_id":"e683b406-530d-4883-8e6f-29049fa30f24","resolution":{"observed_at":"2026-05-18T14:10:15.130757Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-06T17:24:58.979468Z","title":"Internvideo2: Scaling video foundation models for multimodal video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10960","last_updated":"2025-07-15T03:42:14Z","snapshot_observed_at":"2026-08-06T17:17:47.352213Z","submitted_at":"2025-07-15T03:42:14Z","title":"Whom to Respond To? A Transformer-Based Model for Multi-Party Social Robot Interaction","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T17:24:58.979468Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2507.10960"},"observation_digest":"sha256:85964703e778ed59f70dd14c6a4817c9e8cb1c136e9ab8f1059ca83b830c76ae","observation_id":"3f95cab1-e444-4ba3-b563-c49f4618a8e4","resolution":{"observed_at":"2026-08-06T17:24:58.979468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-06T13:57:04.588143Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding.arXiv preprint arXiv:2403.15377, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19924","last_updated":"2025-08-01T12:25:21Z","snapshot_observed_at":"2026-08-07T14:18:30.406157Z","submitted_at":"2025-07-26T12:03:47Z","title":"HumanSAM: Classifying Human-centric Forgery Videos in Human Spatial, Appearance, and Motion Anomaly","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:04.588143Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2507.19924"},"observation_digest":"sha256:b0a8800f3d8104afea45639044e77d2cae59588dcb4e7c0e59b5a57c57f2d824","observation_id":"49664a06-07a5-49c3-a551-aeaea6f87f3f","resolution":{"observed_at":"2026-08-06T13:57:04.588143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-04T20:20:36.886005Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding.arXiv preprint arXiv:2403.15377, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08621","last_updated":"2025-09-10T14:17:53Z","snapshot_observed_at":"2026-08-09T06:09:37.758584Z","submitted_at":"2025-09-10T14:17:53Z","title":"AdsQA: Towards Advertisement Video Understanding","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-04T20:20:36.886005Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2509.08621"},"observation_digest":"sha256:0f5f877f7bf1cc49eda4262a3db01e08254e9600287e6f5fc24eb6f8d0db14dd","observation_id":"63b3f5bf-aa89-4242-8139-ecca80d15c9e","resolution":{"observed_at":"2026-08-04T20:20:36.886005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2510.09608","last_updated":"2025-10-10T17:59:58Z","snapshot_observed_at":"2026-08-08T02:53:06.532732Z","submitted_at":"2025-10-10T17:59:58Z","title":"StreamingVLM: Real-Time Understanding for Infinite Video Streams","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T11:51:33.345812Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2510.09608"},"observation_digest":"sha256:5d8c1bcbd15085eb9af59f7ee0249af1c1ff7e010abcface2180feffd9db0a4e","observation_id":"e4d6e6dc-1985-459c-a7f7-9e3438c90669","resolution":{"observed_at":"2026-05-17T11:51:33.436812Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2604.02891","last_updated":"2026-04-03T09:00:38Z","snapshot_observed_at":"2026-07-06T22:52:10.923214Z","submitted_at":"2026-04-03T09:00:38Z","title":"Progressive Video Condensation with MLLM Agent for Long-form Video Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T20:40:41.380829Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2604.02891"},"observation_digest":"sha256:4f5a64931ace91cc782c35752fe371ffa9538775a22cc67f234729ef3ce9680a","observation_id":"b4991dee-1341-458f-860d-3b3bdb084ac4","resolution":{"observed_at":"2026-05-13T20:43:14.848489Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2605.02834","last_updated":"2026-05-05T09:59:53Z","snapshot_observed_at":"2026-08-07T22:39:16.868213Z","submitted_at":"2026-05-04T17:11:16Z","title":"VideoNet: A Large-Scale Dataset for Domain-Specific Action Recognition","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-08T18:35:48.379198Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2605.02834"},"observation_digest":"sha256:95fa4f988b64ff792054a68a89814daf399641e3d8aed55aa1a72bb7ac050385","observation_id":"9d8ecc9e-0f2c-4a83-b438-04eb4b7f16cc","resolution":{"observed_at":"2026-05-09T06:20:41.219911Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2605.12954","last_updated":"2026-05-13T03:40:21Z","snapshot_observed_at":"2026-07-06T23:24:32.412966Z","submitted_at":"2026-05-13T03:40:21Z","title":"AdaFocus: Adaptive Relevance-Diversity Sampling with Zero-Cache Look-back for Efficient Long Video Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-14T19:43:29.123615Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2605.12954"},"observation_digest":"sha256:eaa88d58a440dfaefe051de3de80a0b7d808de5e76b2532c3b81c944b05b054b","observation_id":"c27af26f-fadf-4167-a8b4-78c00b2f34a4","resolution":{"observed_at":"2026-05-14T19:47:53.818862Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2605.16403","last_updated":"2026-05-13T05:00:19Z","snapshot_observed_at":"2026-07-06T23:27:34.514541Z","submitted_at":"2026-05-13T05:00:19Z","title":"When Vision Speaks for Sound","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-20T22:12:52.160596Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2605.16403"},"observation_digest":"sha256:8efed5da13edb6ef2e70472f7d11b83128debc7449034b06854f70c99c37f6ad","observation_id":"3d550601-08e8-4fef-bd0a-d450e29c01e8","resolution":{"observed_at":"2026-05-20T22:13:46.852107Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2606.00775","last_updated":"2026-05-30T15:40:00Z","snapshot_observed_at":"2026-07-06T23:41:24.911421Z","submitted_at":"2026-05-30T15:40:00Z","title":"GIRL-DETR: Gradient-Isolated Reinforcement Learning for Video Moment Retrieval","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-28T19:12:19.056273Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2606.00775"},"observation_digest":"sha256:6fda4e04f8185b31a27c2fea264c57c5faed4e4a89353c358f7b2d4686594fee","observation_id":"825cb563-f5c1-4de2-8225-143668a3cd13","resolution":{"observed_at":"2026-06-28T19:12:34.375723Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2606.03635","last_updated":"2026-06-02T13:31:57Z","snapshot_observed_at":"2026-08-05T07:22:25.874494Z","submitted_at":"2026-06-02T13:31:57Z","title":"VidMsg: A Benchmark for Implicit Message Inference in Short Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-28T10:25:06.594946Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2606.03635"},"observation_digest":"sha256:118204c1444c9ab899c0c6b36bc598bf5a4b3eb03582fb6f463dac426c024a8b","observation_id":"d949fff9-4f47-4bd9-9828-14cf0d8d1e52","resolution":{"observed_at":"2026-07-02T02:56:30.063976Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2606.06249","last_updated":"2026-06-04T14:52:12Z","snapshot_observed_at":"2026-07-06T23:46:05.829177Z","submitted_at":"2026-06-04T14:52:12Z","title":"GRAMformer: Any-Order Modality Interactions via Volumetric Multimodal Cross-Attention","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T02:22:03.908592Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2606.06249"},"observation_digest":"sha256:d101d5b2025a7a410856bae43f406897f42a55ae425dad95a4049d06092d10c4","observation_id":"d76c4a66-6383-49dc-b854-8df5bd1deb6e","resolution":{"observed_at":"2026-07-02T12:06:56.379175Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:6535416e8af3e2e68be28ba9d6b28e634e17357a600fc00cf26736a831ccf10b","observation_id":"f705bd2b-15de-4e9e-a1c3-f31dd6f8147c","resolution":{"observed_at":"2026-07-04T06:39:37.519902Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2606.28215","last_updated":"2026-06-26T16:05:58Z","snapshot_observed_at":"2026-08-07T08:50:21.361337Z","submitted_at":"2026-06-26T16:05:58Z","title":"HAT-4D: Lifting Monocular Video for 4D Multi-Object Interactions via Human-Agent Collaboration","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-29T04:18:02.341742Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2606.28215"},"observation_digest":"sha256:48c35b483a44987235464721cf2d4b8e7c9818120bb2f3b58d6f0278f636d799","observation_id":"32b0cf74-2c11-4ecd-993b-6de5dc48818f","resolution":{"observed_at":"2026-07-01T17:05:50.568899Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-12T01:50:59.184754Z","title":"arXiv preprint arXiv:2403.15377 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03530","last_updated":"2026-07-03T17:59:58Z","snapshot_observed_at":"2026-08-04T18:44:44.436706Z","submitted_at":"2026-07-03T17:59:58Z","title":"MentalThink: Shaping Thoughts in Mental SVG World","version":1},"reference_index":141,"source":"arxiv_source","source_observed_at":"2026-07-12T01:50:59.184754Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2607.03530"},"observation_digest":"sha256:c542c9e6e5c05c3cb35f59ba4e15b4dff443f4740f44726e92cec408d65dfd04","observation_id":"f73a5272-7603-4603-9b33-02aa4aa36cda","resolution":{"observed_at":"2026-07-12T01:50:59.184754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-01T17:45:04.287534Z","title":"arXiv preprint arXiv:2403.15377 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17560","last_updated":"2026-07-20T05:09:36Z","snapshot_observed_at":"2026-08-05T13:38:45.921055Z","submitted_at":"2026-07-20T05:09:36Z","title":"Reinforcement Learning: From Algorithms To Foundation Models","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-01T17:45:04.287534Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2607.17560"},"observation_digest":"sha256:5fd9d4334188b05fa5af66e34a6b0b59533c7fb4928d83c5e516ae63c888b043","observation_id":"a5be6167-3f15-4bf3-8fd8-5b7bbee536d4","resolution":{"observed_at":"2026-08-01T17:45:04.287534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.15377/citation-record","integrity":"/paper/2403.15377/integrity","json":"/paper/2403.15377/citation-record.json","paper":"/paper/2403.15377"},"outbound":[],"paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 35 inbound Pith citation observations for arXiv:2403.15377."}