{"as_of":"2026-08-13T17:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:45df8dff4173f54466cc6d6cce53e10f8e29aaca8f49a11949070ce150dd23b1","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":22,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":22,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":22,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":22,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T16:57:35.850058Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T12:16:57.715387Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-08-11T16:57:35.850058Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning, 2024a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.09601","last_updated":"2025-03-05T07:06:15Z","snapshot_observed_at":"2026-08-12T09:02:21.582867Z","submitted_at":"2024-12-12T18:59:11Z","title":"TimeRefine: Temporal Grounding with Time Refining Video LLM","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T16:57:35.850058Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2412.09601"},"observation_digest":"sha256:54299fac95e6966eb587cd10d0e10c22b22ecea0d75da21d7865e97796966210","observation_id":"bc418ab9-203b-4948-a194-3c8b11d3d805","resolution":{"observed_at":"2026-08-11T16:57:35.850058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2501.00574","last_updated":"2025-07-13T16:21:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-31T18:01:23Z","title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","version":4},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-18T04:02:43.261543Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2501.00574"},"observation_digest":"sha256:ba1ce8977610335946d2447a67949c218740e0ec053544d2d68d0a568ac1144f","observation_id":"97ab7361-9b0b-40e7-9f20-7f65a83d98de","resolution":{"observed_at":"2026-05-18T04:02:43.587545Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2504.06958","last_updated":"2025-11-11T08:30:00Z","snapshot_observed_at":"2026-08-02T02:31:33.589341Z","submitted_at":"2025-04-09T15:09:27Z","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","version":5},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-15T20:56:07.247122Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2504.06958"},"observation_digest":"sha256:a6eee88fceba81ce1cce0880251fa0d7194b4d7813fd1a731830f35aa185395d","observation_id":"159f87b7-00c8-4b9a-ac7a-9274e28fa6f9","resolution":{"observed_at":"2026-05-15T20:56:07.743281Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-08-07T11:35:47.666720Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning.arXiv preprint arXiv:2410.19702, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01908","last_updated":"2025-06-02T17:28:26Z","snapshot_observed_at":"2026-08-07T11:29:22.385642Z","submitted_at":"2025-06-02T17:28:26Z","title":"Reinforcement Learning Tuning for VideoLLMs: Reward Design and Data Efficiency","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:35:47.666720Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2506.01908"},"observation_digest":"sha256:8f5ccae647eec07bc5581bcc675afee4ffe6c67be411665630caf4497ad13418","observation_id":"997cf40a-d098-4e74-9803-f8343e4597cb","resolution":{"observed_at":"2026-08-07T11:35:47.666720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-08-06T20:29:56.680339Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02591","last_updated":"2025-07-23T07:25:27Z","snapshot_observed_at":"2026-08-09T23:16:14.999128Z","submitted_at":"2025-07-03T12:55:16Z","title":"AuroraLong: Bringing RNNs Back to Efficient Open-Ended Video Understanding","version":3},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-08-06T20:29:56.680339Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2507.02591"},"observation_digest":"sha256:1b464f4ecf0134847c79b5ce22117f03df23cafa85737d8423ec229b34843b60","observation_id":"4e725d3e-281a-4412-bc33-0675a6257fad","resolution":{"observed_at":"2026-08-06T20:29:56.680339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-08-05T22:54:33.231367Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning.arXiv preprint arXiv:2410.19702, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06317","last_updated":"2025-08-08T13:47:00Z","snapshot_observed_at":"2026-08-12T09:06:19.675038Z","submitted_at":"2025-08-08T13:47:00Z","title":"Uncertainty-quantified Rollout Policy Adaptation for Unlabelled Cross-domain Temporal Grounding","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-05T22:54:33.231367Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2508.06317"},"observation_digest":"sha256:d0c4e76ec882baa462a6d18f5c9c4700e9538bdad9955f027737f67eec560839","observation_id":"e3ee6162-9ba8-4623-a6bd-2d0f5db2f5a9","resolution":{"observed_at":"2026-08-05T22:54:33.231367Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-08-04T07:23:08.354942Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.26113","last_updated":"2026-06-18T20:06:47Z","snapshot_observed_at":"2026-08-06T20:27:36.141566Z","submitted_at":"2025-10-30T03:53:22Z","title":"EgoExo-Con: Exploring View-Invariant Video Temporal Understanding","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-04T07:23:08.354942Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2510.26113"},"observation_digest":"sha256:563f673333e06008b00e4edd64f719afb23d748d69a38573c66a1fe9fd42d6da","observation_id":"a5502050-69e0-428c-8a3b-03e29de7de9a","resolution":{"observed_at":"2026-08-04T07:23:08.354942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-08-03T20:23:09.387510Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning.arXiv preprint arXiv:2410.19702, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.20272","last_updated":"2026-07-03T06:27:17Z","snapshot_observed_at":"2026-08-13T06:51:54.672676Z","submitted_at":"2025-11-25T12:58:32Z","title":"VKnowU: Evaluating Visual Knowledge Understanding in Multimodal LLMs","version":2},"reference_index":113,"source":"pdf_text","source_observed_at":"2026-08-03T20:23:09.387510Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2511.20272"},"observation_digest":"sha256:7500ae8f556316982acc35eb5d285e43e41a64eff29ce775f79b53f4068a93bc","observation_id":"5c3f8b57-a5f9-4bc9-898f-fcaf12ce948a","resolution":{"observed_at":"2026-08-03T20:23:09.387510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2512.03043","last_updated":"2026-04-28T12:07:36Z","snapshot_observed_at":"2026-08-13T15:37:47.305433Z","submitted_at":"2025-12-02T18:59:52Z","title":"OneThinker: All-in-one Reasoning Model for Image and Video","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-17T02:09:39.820651Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2512.03043"},"observation_digest":"sha256:1065656477c7121ffc5ca1465518c5fcd60695916a70aa4808c86588633ccc0f","observation_id":"baa7b7fe-03bb-40a5-a8a2-0fb24e0bece8","resolution":{"observed_at":"2026-05-17T02:11:26.627257Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2512.03963","last_updated":"2026-04-14T11:28:58Z","snapshot_observed_at":"2026-08-13T01:36:01.137921Z","submitted_at":"2025-12-03T16:57:00Z","title":"TempR1: Improving Temporal Understanding of MLLMs via Temporal-Aware Multi-Task Reinforcement Learning","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-17T02:18:21.718091Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2512.03963"},"observation_digest":"sha256:10169e22f81c6941174caa971b614d692471df85d82dfa7b1d758261b3f91787","observation_id":"e0fb155f-e300-4ee5-a867-e80b8a875f35","resolution":{"observed_at":"2026-05-17T02:18:52.310729Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2602.02994","last_updated":"2026-06-02T08:33:06Z","snapshot_observed_at":"2026-08-11T10:27:58.772493Z","submitted_at":"2026-02-03T02:05:48Z","title":"Video-OPD: Efficient Post-Training of Multimodal Large Language Models for Temporal Video Grounding via On-Policy Distillation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-16T08:38:49.075457Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2602.02994"},"observation_digest":"sha256:def85636bbeb14a16f91ec9f788ed2f9d09446eaa84071d08a71d9e02b8956e3","observation_id":"6ddec4d3-95ee-48a5-a744-89f4e1e3e8da","resolution":{"observed_at":"2026-05-16T08:40:46.392368Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-08-03T05:14:24.676023Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02994","last_updated":"2026-06-02T08:33:06Z","snapshot_observed_at":"2026-08-11T10:27:58.772493Z","submitted_at":"2026-02-03T02:05:48Z","title":"Video-OPD: Efficient Post-Training of Multimodal Large Language Models for Temporal Video Grounding via On-Policy Distillation","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-03T05:14:24.676023Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2602.02994"},"observation_digest":"sha256:a3c668a90ea57fcbfa0bd4732871e0191485f75a3c6e1430f6f0c33d90a1f7ab","observation_id":"efbf7782-1266-499a-bde6-1ef39343e1c2","resolution":{"observed_at":"2026-08-03T05:14:24.676023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2604.08014","last_updated":"2026-04-21T06:24:07Z","snapshot_observed_at":"2026-08-12T18:01:28.612956Z","submitted_at":"2026-04-09T09:14:00Z","title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","version":3},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:16.204012Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2604.08014"},"observation_digest":"sha256:1a6578de49b5dd977fab57f94f2ab4d6e3ee72eab4a99485db1ee8522d8f456c","observation_id":"616db61f-0708-4d9d-b4f9-5467b24966ae","resolution":{"observed_at":"2026-05-11T00:15:52.911593Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2604.09037","last_updated":"2026-04-10T06:58:29Z","snapshot_observed_at":"2026-08-13T15:29:00.642103Z","submitted_at":"2026-04-10T06:58:29Z","title":"SiMing-Bench: Evaluating Procedural Correctness from Continuous Interactions in Clinical Skill Videos","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-10T17:37:40.373211Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2604.09037"},"observation_digest":"sha256:467f81cd189c074a25ea1ec733019618e507ccdb71bbed4d2c6d654cfddfa03a","observation_id":"dc5bf10c-afc4-4660-8df9-438a6e168374","resolution":{"observed_at":"2026-05-11T06:30:58.457792Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2604.25886","last_updated":"2026-06-16T05:05:15Z","snapshot_observed_at":"2026-08-08T01:22:27.799880Z","submitted_at":"2026-04-28T17:29:19Z","title":"MarkIt: Training-Free Visual Markers for Precise Video Temporal Grounding","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-07T14:10:27.416341Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2604.25886"},"observation_digest":"sha256:bee2076051f616ea2429167385f9eb4f2a59123ba42e3d3c90838004c89d6dae","observation_id":"80dfd7e6-4987-4bb0-92f2-c69d73083948","resolution":{"observed_at":"2026-05-12T00:46:13.722151Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2604.25886","last_updated":"2026-06-16T05:05:15Z","snapshot_observed_at":"2026-08-08T01:22:27.799880Z","submitted_at":"2026-04-28T17:29:19Z","title":"MarkIt: Training-Free Visual Markers for Precise Video Temporal Grounding","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-01T08:41:31.700367Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2604.25886"},"observation_digest":"sha256:5ad93b624e24660ab6ee9ae3010a05658ddc68f3e9c4a8dce6e24bbc5c3a111f","observation_id":"f81d4d1d-90c9-4b7f-b36d-96c878701b03","resolution":{"observed_at":"2026-07-01T08:45:34.796582Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2605.13803","last_updated":"2026-05-13T17:25:51Z","snapshot_observed_at":"2026-08-11T00:27:38.038120Z","submitted_at":"2026-05-13T17:25:51Z","title":"EvoGround: Self-Evolving Video Agents for Video Temporal Grounding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-14T19:29:47.356665Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2605.13803"},"observation_digest":"sha256:b5a953c0c1ca33ef03e7ae9ed0e727178756b3dc6ab7b7a8b343015f4d2035ad","observation_id":"dec28f3d-ec1d-4d27-ae77-e5c6a20a027b","resolution":{"observed_at":"2026-05-14T19:32:52.475211Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2605.14733","last_updated":"2026-05-14T11:56:14Z","snapshot_observed_at":"2026-08-06T01:04:24.065010Z","submitted_at":"2026-05-14T11:56:14Z","title":"Video-Zero: Self-Evolution Video Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-30T21:32:16.939563Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2605.14733"},"observation_digest":"sha256:3e98c2041e01ea6844147933d1d74a51cdb9fc027f5157518846fa76fcb84196","observation_id":"212aca7d-9bd8-4f19-b8f6-149238165a3d","resolution":{"observed_at":"2026-06-30T21:35:04.432794Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2605.21954","last_updated":"2026-05-21T03:40:22Z","snapshot_observed_at":"2026-07-06T23:32:20.708664Z","submitted_at":"2026-05-21T03:40:22Z","title":"MLLMs Know When Before Speaking: Revealing and Recovering Temporal Grounding via Attention Cues","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-22T07:13:43.716510Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2605.21954"},"observation_digest":"sha256:06e3afda117cf877f078bdb1784b5ae5495c0cd0e19b7bc2c6943ed9d986aa7d","observation_id":"baee4459-ff62-4346-ac17-0a7b67998065","resolution":{"observed_at":"2026-05-22T07:14:42.429049Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":"2410.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-07-02T12:16:57.715387Z","title":"Timesuite: Improving mllms for long video understanding via grounded tuning","venue":null,"work_id":"f2ca8961-3931-4220-a78a-7f084312f7b0","year":2024},"citing_paper":{"arxiv_id":"2606.06294","last_updated":"2026-06-21T07:27:04Z","snapshot_observed_at":"2026-08-13T09:35:12.990364Z","submitted_at":"2026-06-04T15:31:22Z","title":"Towards One-to-Many Temporal Grounding","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-06-28T02:11:48.455492Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2606.06294"},"observation_digest":"sha256:eba8b65cdb49486b9b758a0bdc1ead4864e4872637f4550632d5f2181c8db5c0","observation_id":"b6449e2b-3713-43c5-8c3a-1ec12e3ba111","resolution":{"observed_at":"2026-07-02T12:16:57.717129Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-08-01T14:37:48.193803Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.18716","last_updated":"2026-07-21T05:16:36Z","snapshot_observed_at":"2026-08-11T19:04:06.532396Z","submitted_at":"2026-07-21T05:16:36Z","title":"Continual Video-MLLM Adaptation over Evolving Domains","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-01T14:37:48.193803Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2607.18716"},"observation_digest":"sha256:2f57d873a742f5d793ac5defb33380f4c2f883acf9615394a7b1ea17fb3e0be2","observation_id":"09fdb61b-9e31-405b-a733-5435a63d754c","resolution":{"observed_at":"2026-08-01T14:37:48.193803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19702","snapshot_observed_at":"2026-08-04T17:24:16.112662Z","title":"2410.19702 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01980","last_updated":"2026-08-03T09:42:42Z","snapshot_observed_at":"2026-08-11T15:40:04.127681Z","submitted_at":"2026-08-03T09:42:42Z","title":"AdaThinkV: Adaptive Thinking for Token-Efficient Video Reasoning","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-04T17:24:16.112662Z"},"links":{"cited_paper":"/paper/2410.19702","citing_paper":"/paper/2608.01980"},"observation_digest":"sha256:3f89b5eb0f4f94776096d8f0ffc6d9f1307e5ad96b5cbc44d63371b2ce60fe01","observation_id":"bd566c8d-0625-48e0-b6fd-950c062b82fc","resolution":{"observed_at":"2026-08-04T17:24:16.112662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2410.19702/citation-record","integrity":"/paper/2410.19702/integrity","json":"/paper/2410.19702/citation-record.json","paper":"/paper/2410.19702"},"outbound":[],"paper":{"arxiv_id":"2410.19702","last_updated":"2025-02-12T16:47:30Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-13T10:15:48.958870Z","submitted_at":"2024-10-25T17:19:55Z","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 22 inbound Pith citation observations for arXiv:2410.19702."}