{"as_of":"2026-08-12T07:39:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9c98c381ec4452aacfe4e4112c125e34cf0b7c655e766443cb9797e93abb36c2","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":7,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":7,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T13:44:50.317635Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T10:29:44.913681Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-08-11T13:44:50.317635Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12833","last_updated":"2025-05-28T09:22:24Z","snapshot_observed_at":"2026-08-11T17:45:15.774121Z","submitted_at":"2024-12-17T11:54:47Z","title":"FocusChat: Text-guided Long Video Understanding via Spatiotemporal Information Filtering","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T13:44:50.317635Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2412.12833"},"observation_digest":"sha256:d534f55f8bf0d17f0a297427fca558c7ec8c620d0e026b64949791e104e1b407","observation_id":"b4cf9ec3-f3d7-4196-b36c-00a50127b88e","resolution":{"observed_at":"2026-08-11T13:44:50.317635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-08-06T15:53:32.306216Z","title":"Ranasinghe, X","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14784","last_updated":"2025-08-18T09:06:46Z","snapshot_observed_at":"2026-08-09T23:46:47.099823Z","submitted_at":"2025-07-20T01:57:00Z","title":"LeAdQA: LLM-Driven Context-Aware Temporal Grounding for Video Question Answering","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T15:53:32.306216Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2507.14784"},"observation_digest":"sha256:b67f3895fda1903b1248759a3d0dc08f00c2a4af8c67b83ba2b8d9d4ad4429b4","observation_id":"d6ee249b-7f99-4517-b17d-92e41ccbddc6","resolution":{"observed_at":"2026-08-06T15:53:32.306216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":"2403.16998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-07-04T10:29:44.913681Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":"a0d8f834-29b0-4597-ad47-382843695ca9","year":2024},"citing_paper":{"arxiv_id":"2604.02891","last_updated":"2026-04-03T09:00:38Z","snapshot_observed_at":"2026-08-11T16:19:37.738972Z","submitted_at":"2026-04-03T09:00:38Z","title":"Progressive Video Condensation with MLLM Agent for Long-form Video Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T20:40:41.380829Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2604.02891"},"observation_digest":"sha256:1df589c57f5b70ba031c0250d869b1b6a25d3bfd4f19317dec70ee55ff31a4c7","observation_id":"c930cb75-2276-44ae-adab-2cf010a1b749","resolution":{"observed_at":"2026-05-13T20:43:14.871880Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":"2403.16998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-07-04T10:29:44.913681Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":"a0d8f834-29b0-4597-ad47-382843695ca9","year":2024},"citing_paper":{"arxiv_id":"2604.14692","last_updated":"2026-05-15T12:00:53Z","snapshot_observed_at":"2026-08-11T16:10:49.896942Z","submitted_at":"2026-04-16T06:50:20Z","title":"Chain-of-Glimpse: Search-Guided Progressive Object-Grounded Reasoning for Video Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T12:03:09.408019Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2604.14692"},"observation_digest":"sha256:5e5fa66a602f243fe22851ba06727beba4156645cf307339f8a79572c820b051","observation_id":"aa8772c6-d631-472d-b460-9edf5730b0a3","resolution":{"observed_at":"2026-05-10T12:05:22.177953Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":"2403.16998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-07-04T10:29:44.913681Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":"a0d8f834-29b0-4597-ad47-382843695ca9","year":2024},"citing_paper":{"arxiv_id":"2604.14692","last_updated":"2026-05-15T12:00:53Z","snapshot_observed_at":"2026-08-11T16:10:49.896942Z","submitted_at":"2026-04-16T06:50:20Z","title":"Chain-of-Glimpse: Search-Guided Progressive Object-Grounded Reasoning for Video Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-19T17:34:10.344111Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2604.14692"},"observation_digest":"sha256:38f42d896895c8d89a416b5d2e6f13dcd104f3e157596b7065b0557b06cfb78b","observation_id":"ac4c87ab-087c-4c2d-8458-7591a1866ac2","resolution":{"observed_at":"2026-05-19T17:37:41.750865Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":"2403.16998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-07-04T10:29:44.913681Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":"a0d8f834-29b0-4597-ad47-382843695ca9","year":2024},"citing_paper":{"arxiv_id":"2605.08974","last_updated":"2026-05-09T14:32:36Z","snapshot_observed_at":"2026-08-11T16:09:05.826725Z","submitted_at":"2026-05-09T14:32:36Z","title":"Tracking the Truth: Object-Centric Spatio-Temporal Monitoring for Video Large Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-12T02:31:40.463891Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2605.08974"},"observation_digest":"sha256:e241bfe815830b6167beb19e1fa75246ab30d3eac0ed0db01937f13d5f2ef4fb","observation_id":"9f09208b-d686-492b-96d3-8d3ef94bd99a","resolution":{"observed_at":"2026-05-12T07:36:31.182712Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":"2403.16998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-07-04T10:29:44.913681Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":"a0d8f834-29b0-4597-ad47-382843695ca9","year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:005b2bf0ebe56c68873cb5d6a1232c3a59e12bb4a620b421431e33eea5987f0a","observation_id":"0a58456e-931f-4ce1-a837-dae08856644d","resolution":{"observed_at":"2026-07-04T10:29:44.915306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2403.16998/citation-record","integrity":"/paper/2403.16998/integrity","json":"/paper/2403.16998/citation-record.json","paper":"/paper/2403.16998"},"outbound":[],"paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","latest_version":5,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 7 inbound Pith citation observations for arXiv:2403.16998."}