{"as_of":"2026-08-10T07:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:43f697ac664ee0d836b9629a4da3652cf97b9897ce7a19e0752452ea371449ea","coverage":[{"denominator":33,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:22:38.056203Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T04:26:51.521243Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T11:28:04.115693Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"cited_work":{"arxiv_id":"2506.21873","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.21873","snapshot_observed_at":"2026-07-03T11:28:04.115693Z","title":"Grounding-aware token pruning: Recovering from drastic performance drops in visual grounding caused by pruning, 2025","venue":null,"work_id":"2e1fc42d-eac9-4215-b5ea-a3c62cfc02f8","year":2025},"citing_paper":{"arxiv_id":"2606.12412","last_updated":"2026-06-10T17:59:57Z","snapshot_observed_at":"2026-08-03T05:45:43.148184Z","submitted_at":"2026-06-10T17:59:57Z","title":"Reroute, Don't Remove: Recoverable Visual Token Routing for Vision-Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T09:35:24.118536Z"},"links":{"cited_paper":"/paper/2506.21873","citing_paper":"/paper/2606.12412"},"observation_digest":"sha256:30e6981db0c17f45deaa06e3755d291383d0185190a2e9e1b7f49864780d751b","observation_id":"1d10623c-83fb-4f45-95cd-23b3146a6d3d","resolution":{"observed_at":"2026-07-03T11:28:04.117867Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"cited_work":{"arxiv_id":"2506.21873","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.21873","snapshot_observed_at":"2026-07-03T11:28:04.115693Z","title":"Grounding-aware token pruning: Recovering from drastic performance drops in visual grounding caused by pruning, 2025","venue":null,"work_id":"2e1fc42d-eac9-4215-b5ea-a3c62cfc02f8","year":2025},"citing_paper":{"arxiv_id":"2606.31599","last_updated":"2026-06-30T12:47:30Z","snapshot_observed_at":"2026-07-07T00:05:17.259491Z","submitted_at":"2026-06-30T12:47:30Z","title":"Token-Sparse Medical Multimodal Reasoning via Dual-Stream Reinforcement Learning","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-07-01T05:36:43.609602Z"},"links":{"cited_paper":"/paper/2506.21873","citing_paper":"/paper/2606.31599"},"observation_digest":"sha256:75727e584f06ea858d8eec00fa4557859a14c925c943bb6ca052ed0bb0ea0814","observation_id":"dd37fb2c-b09e-4a78-b543-4a85336b2d2f","resolution":{"observed_at":"2026-07-01T10:15:45.167041Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.21873","snapshot_observed_at":"2026-08-04T01:29:54.348125Z","title":"arXiv preprint arXiv:2506.21873 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00077","last_updated":"2026-08-04T10:46:13Z","snapshot_observed_at":"2026-08-07T23:11:47.320726Z","submitted_at":"2026-07-29T13:45:31Z","title":"Beyond Accuracy: Auditing Spatial Provenance in Visual Token Pruning for OCR-Critical MLLM Inference","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-04T01:29:54.348125Z"},"links":{"cited_paper":"/paper/2506.21873","citing_paper":"/paper/2608.00077"},"observation_digest":"sha256:1f774a17c68d7d40da6f5d8bc4dfea51e9e0feb5415c3c78986e85283b62964d","observation_id":"9580cabb-5320-4a7c-ba02-205e08d07265","resolution":{"observed_at":"2026-08-04T01:29:54.348125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.21873","snapshot_observed_at":"2026-08-05T04:26:51.521243Z","title":"arXiv preprint arXiv:2506.21873 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00077","last_updated":"2026-08-04T10:46:13Z","snapshot_observed_at":"2026-08-07T23:11:47.320726Z","submitted_at":"2026-07-29T13:45:31Z","title":"Beyond Accuracy: Auditing Spatial Provenance in Visual Token Pruning for OCR-Critical MLLM Inference","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-05T04:26:51.521243Z"},"links":{"cited_paper":"/paper/2506.21873","citing_paper":"/paper/2608.00077"},"observation_digest":"sha256:16987e6af8caca48c81ac12aea325b960cfb136bffc9d402a09c06c9f3de10eb","observation_id":"1208ac5a-4bec-4880-8f3e-70d05e81322a","resolution":{"observed_at":"2026-08-05T04:26:51.521243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.21873/citation-record","integrity":"/paper/2506.21873/integrity","json":"/paper/2506.21873/citation-record.json","paper":"/paper/2506.21873"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2103.00020","last_updated":"2021-02-26T19:04:58Z","snapshot_observed_at":"2026-07-06T10:45:03.059688Z","submitted_at":"2021-02-26T19:04:58Z","title":"Learning Transferable Visual Models From Natural Language Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.00020","snapshot_observed_at":"2026-08-06T22:22:34.222152Z","title":"Learning transferable visual models from natu- ral language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:34.222152Z"},"links":{"cited_paper":"/paper/2103.00020","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:b5ba69ed43b675781717b3df555add420cacdb3273b1b19beb013a1b1a375234","observation_id":"87ff0b8a-ca3b-4887-b17f-c9c0cec093d5","resolution":{"observed_at":"2026-08-06T22:22:34.222152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:40.719527Z","title":"Lmms-eval: Accelerating the development of large multimoal models, 2024","venue":null,"work_id":"99a46a9f-19b5-4402-93d7-9a8778506457","year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:34.319536Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:f0bb0acaa064b04189bf3a6825f8720c6a3fa6489032bb86879fea9f244671ae","observation_id":"6a4d00e1-d3d4-4eff-8769-671d4229e0c0","resolution":{"observed_at":"2026-08-06T22:22:40.797436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:34.453915Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:34.453915Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:11ef1b6cc682414ef9387fb211555049d752b729d3a238561cc03d174555a42f","observation_id":"13662c94-8ae8-41ee-ae62-f8865c5086aa","resolution":{"observed_at":"2026-08-06T22:22:34.453915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:40.565158Z","title":"PuMer: Pruning and merging tokens for efficient vision language models","venue":null,"work_id":"43aae28a-9e85-4547-8cdd-15055b81257e","year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:34.502806Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:45b9c0548ccbd8e67c67b3f9fd4e47f507f838d3109e2be2eb1fb7a4d71d1104","observation_id":"ed13873d-7303-41c2-887f-f70ad36352f5","resolution":{"observed_at":"2026-08-06T22:22:40.638759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15195","last_updated":"2023-07-03T16:08:00Z","snapshot_observed_at":"2026-07-06T15:47:07.545213Z","submitted_at":"2023-06-27T04:31:52Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15195","snapshot_observed_at":"2026-08-06T22:22:34.600551Z","title":"Shikra: Unleashing multi- modal llm’s referential dialogue magic","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:34.600551Z"},"links":{"cited_paper":"/paper/2306.15195","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:5f5275ead7deb5f62e758816c64d6071d3622f082de05437da6137bf413ff041","observation_id":"3f9feba8-b212-40bb-b00d-6e25488e19e9","resolution":{"observed_at":"2026-08-06T22:22:34.600551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:40.406044Z","title":"Chasing sparsity in vision transform- ers: An end-to-end exploration","venue":null,"work_id":"0cd49533-3beb-4872-abb5-283707a491fa","year":2021},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:36.588326Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:1e91f1102f9337a87fe93b6962f5c8dbe215fa206ea72c3f3fa575625fdc2d8b","observation_id":"5af470b7-381b-4a2d-9d38-e97d8fc76450","resolution":{"observed_at":"2026-08-06T22:22:40.493296Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01179","last_updated":"2024-12-19T06:26:04Z","snapshot_observed_at":"2026-07-06T19:09:14.628371Z","submitted_at":"2024-09-02T11:19:54Z","title":"Recoverable Compression: A Multimodal Vision Token Recovery Mechanism Guided by Text Information","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01179","snapshot_observed_at":"2026-08-06T22:22:36.892337Z","title":"Recoverable compression: A mul- timodal vision token recovery mechanism guided by text in- formation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:36.892337Z"},"links":{"cited_paper":"/paper/2409.01179","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:f3c926e7a14a22426298ccc01a49ebddc91eb19b41c9ccabcb32f8cafb9cfa7a","observation_id":"7977b23f-3ab8-4e3b-a190-dbfb2a7e998a","resolution":{"observed_at":"2026-08-06T22:22:36.892337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:40.226085Z","title":"Bert: Pre-training of deep bidirectional trans- formers for language understanding, 2019","venue":null,"work_id":"9b33d571-8167-4275-9463-39d6415fb6e3","year":2019},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:36.947010Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:0b329d424b1eb7777b1ebc155931ae403476f304b0650a6fdcc177e2ed771fe0","observation_id":"fc0b8107-ea83-4856-9830-afdc21830e16","resolution":{"observed_at":"2026-08-06T22:22:40.317527Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-06T22:22:37.063835Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.063835Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:61b2a88ce7c116a95d57ecb895c64c9787450a5b91c09801be1162d9dbb7ee99","observation_id":"9d0203be-a9ae-4267-b819-47f2116e220d","resolution":{"observed_at":"2026-08-06T22:22:37.063835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:40.051463Z","title":"Mme: A compre- hensive evaluation benchmark for multimodal large language models, 2024","venue":null,"work_id":"591fe45c-7102-495c-82a7-79ab4a5547af","year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.158952Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:f29c1184108730998502e4f99a28c64efd32349649260ce986410c147458d09b","observation_id":"80d3a7ef-ddd5-48d5-8a6d-db11f437122b","resolution":{"observed_at":"2026-08-06T22:22:40.112200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:39.845194Z","title":"Stangl, Anhong Guo, Chi Lin, Kristen Grauman, Jiebo Luo, and Jeffrey P","venue":null,"work_id":"7f038059-1292-4ccb-bf8b-35e1e00d4ae6","year":2018},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.240849Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:a2cea1b26b62b0ca1d19453c03acf2da8c18d414be6609351da90eab95253550","observation_id":"6252318a-6096-4ab7-8ba4-8f2660f663eb","resolution":{"observed_at":"2026-08-06T22:22:39.929819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07704","last_updated":"2023-10-11T17:55:15Z","snapshot_observed_at":"2026-07-06T16:31:25.350087Z","submitted_at":"2023-10-11T17:55:15Z","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07704","snapshot_observed_at":"2026-08-06T22:22:37.288321Z","title":"Ferret: Refer and ground anything anywhere at any granularity","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.288321Z"},"links":{"cited_paper":"/paper/2310.07704","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:a8ee99e86100744f828156c545a53ec64cbef3a61bd22ab8c6ecceeed1451fc4","observation_id":"96308886-7034-4e99-b06c-ec9193ca46fc","resolution":{"observed_at":"2026-08-06T22:22:37.288321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:39.654091Z","title":"Hudson and Christopher D","venue":null,"work_id":"7b66f5e5-cf26-4379-9ede-22b3514985dc","year":2019},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.322374Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:b500edce09d48352989aa479146aee52489d5df836d1517c1e05c7c54c90cec9","observation_id":"94758521-e0ad-4fef-9396-4cb57a0b7a2c","resolution":{"observed_at":"2026-08-06T22:22:39.759716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.09478","last_updated":"2023-11-07T18:25:48Z","snapshot_observed_at":"2026-08-06T12:38:03.720232Z","submitted_at":"2023-10-14T03:22:07Z","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.09478","snapshot_observed_at":"2026-08-06T22:22:37.354640Z","title":"Minigpt- v2: Large language model as a unified interface for vision- language multi-task learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.354640Z"},"links":{"cited_paper":"/paper/2310.09478","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:63236b6720b55e16fad001d65320c6a703e347ffb8c52b6e8b2dc45f0abc9524","observation_id":"7ffded63-3567-47ec-910b-4f75f5548ef0","resolution":{"observed_at":"2026-08-06T22:22:37.354640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.12763","last_updated":"2021-10-12T00:49:54Z","snapshot_observed_at":"2026-08-03T17:40:24.925899Z","submitted_at":"2021-04-26T17:55:33Z","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.12763","snapshot_observed_at":"2026-08-06T22:22:37.398957Z","title":"Mdetr– modulated detection for end-to-end multi-modal understand- ing","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.398957Z"},"links":{"cited_paper":"/paper/2104.12763","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:b690be740ea897ccb1d883260995fe1c8726dcaeae5bf9d963a7d90420340dda","observation_id":"e5177977-f620-44b3-af16-da781b15a257","resolution":{"observed_at":"2026-08-06T22:22:37.398957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:39.492458Z","title":"ReferItGame: Referring to objects in pho- tographs of natural scenes","venue":null,"work_id":"83f6d142-9d5b-4e20-88d7-95a80ba9b511","year":2014},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.439860Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:1e465812329e9713e2ba5007b7762dc36ce8f68adc2ffc707e7eacf353660593","observation_id":"1047d42b-4035-4686-8f2f-34b0c62f05e4","resolution":{"observed_at":"2026-08-06T22:22:39.581887Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-06T22:22:37.486139Z","title":"Llava-onevision: Easy visual task transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.486139Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:d603f7526105c857f46e4ca17a504738f8921e3819136ae94894b658b5486b08","observation_id":"6aaf2107-1780-41a2-9582-3bfcd228cdca","resolution":{"observed_at":"2026-08-06T22:22:37.486139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:39.300918Z","title":"Improved baselines with visual instruction tuning, 2023","venue":null,"work_id":"7b0014f3-3bee-4c7b-92eb-399dff8e7a2a","year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.532926Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:636a2844a6d4efca64f1d081e1e310d6314c6bb1aeda936dd3b3e230afee7d18","observation_id":"6b850b5a-68b8-422d-8942-b13c090b933f","resolution":{"observed_at":"2026-08-06T22:22:39.392107Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:39.124690Z","title":"Visual instruction tuning","venue":null,"work_id":"6a7aaff5-c77a-461b-a2cc-5d8239b5f914","year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.566606Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:afd90bcafb6be8043be739d973565f5a4cff3b6c2f48b63a861b2e8cda5fb3ef","observation_id":"68059e97-b956-4043-b2dd-cd8c386fdc96","resolution":{"observed_at":"2026-08-06T22:22:39.211131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:38.986948Z","title":"Llava-next: Im- proved reasoning, ocr, and world knowledge, 2024","venue":null,"work_id":"5e965e34-1315-4503-a7a6-b4899a310554","year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.595557Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:ed3a7733db5e19e6ac83a995415233d5676d227229cf0b507f3387607061e755","observation_id":"a98f1f51-a427-4ee7-8008-049e208faa8f","resolution":{"observed_at":"2026-08-06T22:22:39.048593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:38.776008Z","title":"Ok-vqa: A visual question answer- ing benchmark requiring external knowledge","venue":null,"work_id":"d271bd48-0ca9-4e18-b137-00a6a6fc3489","year":2019},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.636569Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:d7b1c3c10c2fcbcfbdf5ae4e543916f87042febb3f2c50a3d7dbee92eecc5887","observation_id":"56f3c6af-f16a-4ade-a470-9398eb3fad83","resolution":{"observed_at":"2026-08-06T22:22:38.868981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:37.656404Z","title":"Llava-prumerge: Adaptive token reduction for efficient large multimodal models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.656404Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:7797db60150850ed46540ece7150892e56b7f4fd4071c9775788e8c0f0102017","observation_id":"54baaea2-b91c-4979-b628-4e52c11da981","resolution":{"observed_at":"2026-08-06T22:22:37.656404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-06T22:22:37.694996Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.694996Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:f5c3d7d5eed5a3339700f007e3f2d2f867e8bc765eecfa8eecaca9d3be036f51","observation_id":"503b3c1d-7e40-4cf9-b445-c27e3b6e1c85","resolution":{"observed_at":"2026-08-06T22:22:37.694996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.10994","last_updated":"2024-12-17T02:05:27Z","snapshot_observed_at":"2026-07-06T19:16:36.688367Z","submitted_at":"2024-09-17T08:56:27Z","title":"Less is More: A Simple yet Effective Token Reduction Method for Efficient Multi-modal LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.10994","snapshot_observed_at":"2026-08-06T22:22:37.715782Z","title":"Less is more: A sim- ple yet effective token reduction method for efficient multi- modal llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.715782Z"},"links":{"cited_paper":"/paper/2409.10994","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:30a231174b2883e73a6f2c4729a88e29d6957e04eecabaf1f50eee50a0d88b7e","observation_id":"8632f363-36a1-41db-8844-8a1c699956a1","resolution":{"observed_at":"2026-08-06T22:22:37.715782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.15033","last_updated":"2024-02-26T15:26:20Z","snapshot_observed_at":"2026-08-05T16:55:19.542695Z","submitted_at":"2023-05-24T11:18:00Z","title":"SmartTrim: Adaptive Tokens and Attention Pruning for Efficient Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2305.15033","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.15033","snapshot_observed_at":"2026-08-06T22:22:38.192026Z","title":"SmartTrim: Adaptive Tokens and Attention Pruning for Efficient Vision-Language Models","venue":"cs.CL","work_id":"c57fed02-5dc9-471f-afe7-8e1d85f74346","year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.743642Z"},"links":{"cited_paper":"/paper/2305.15033","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:c23651e6490bc8ee5ab2998f69fa7178a1ddfe450ce23610bb7346a4a31ff3b5","observation_id":"7f131960-fa44-40a5-9d6e-9b4f5b04fd33","resolution":{"observed_at":"2026-08-06T22:22:38.253638Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.03079","last_updated":"2024-02-04T08:23:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-06T13:04:39Z","title":"CogVLM: Visual Expert for Pretrained Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.03079","snapshot_observed_at":"2026-08-06T22:22:37.779534Z","title":"Cogvlm: Visual expert for pretrained language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.779534Z"},"links":{"cited_paper":"/paper/2311.03079","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:a97dd1f8d7f69a8c7d9597a3fe34c69eb156f28d5a00ae8245464d6fba240f5c","observation_id":"9006af8c-53da-4f81-ac21-12430f38a059","resolution":{"observed_at":"2026-08-06T22:22:37.779534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15343","last_updated":"2023-09-27T12:05:41Z","snapshot_observed_at":"2026-07-06T15:08:30.190912Z","submitted_at":"2023-03-27T15:53:01Z","title":"Sigmoid Loss for Language Image Pre-Training","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.15343","snapshot_observed_at":"2026-08-06T22:22:37.804075Z","title":"Sigmoid loss for language image pre- training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.804075Z"},"links":{"cited_paper":"/paper/2303.15343","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:84317c2a14e1b5ed4ffadbe9f578d3a1734e946d49390375975304c1f8b18dad","observation_id":"740a8bef-e1ff-46b1-89b5-71d8a266bd68","resolution":{"observed_at":"2026-08-06T22:22:37.804075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.09554","last_updated":"2020-12-07T04:56:24Z","snapshot_observed_at":"2026-07-06T09:39:55.355752Z","submitted_at":"2020-07-19T01:45:02Z","title":"Referring Expression Comprehension: A Survey of Methods and Datasets","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.09554","snapshot_observed_at":"2026-08-06T22:22:37.839061Z","title":"Referring expression comprehension: A survey of methods and datasets","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.839061Z"},"links":{"cited_paper":"/paper/2007.09554","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:76b82191128747fc25fdbbded3de414bdbbb8e206d2c9a126be0e01b12debd32","observation_id":"f0abc5a8-7da7-4a9b-b116-4f52f17b07b3","resolution":{"observed_at":"2026-08-06T22:22:37.839061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01162","last_updated":"2025-02-12T04:35:28Z","snapshot_observed_at":"2026-07-06T19:09:14.628371Z","submitted_at":"2024-09-02T10:49:10Z","title":"Sparsity Meets Similarity: Leveraging Long-Tail Distribution for Dynamic Optimized Token Representation in Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01162","snapshot_observed_at":"2026-08-06T22:22:37.891208Z","title":"Balancing performance and efficiency: A multimodal large language model pruning method based image text interaction.ArXiv, abs/2409.01162,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.891208Z"},"links":{"cited_paper":"/paper/2409.01162","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:d451cfcb3c3a0660cfbe49217a013def39ea9f0a9cd29ac496bb31dc5544dfe8","observation_id":"00fd528a-580c-476a-9dff-37e14e2d1699","resolution":{"observed_at":"2026-08-06T22:22:37.891208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:38.627571Z","title":"Mm-vet: Evaluating large multimodal models for integrated capabilities, 2023","venue":null,"work_id":"f469aaa2-cde0-4d86-a9cf-fd5c4a90ee8f","year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.932357Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:afe27e5225839d5613f6466af11444169779f55b08998a2fe1fa75caea6f9e61","observation_id":"f291d98b-bdaa-49bb-bf10-0afed6d4abc4","resolution":{"observed_at":"2026-08-06T22:22:38.683766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.07636","last_updated":"2022-12-05T13:53:51Z","snapshot_observed_at":"2026-07-06T14:18:10.647862Z","submitted_at":"2022-11-14T18:59:52Z","title":"EVA: Exploring the Limits of Masked Visual Representation Learning at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.07636","snapshot_observed_at":"2026-08-06T22:22:37.974207Z","title":"Eva: Exploring the limits of masked visual representation learning at scale","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.974207Z"},"links":{"cited_paper":"/paper/2211.07636","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:49458bb6ab8a02cf577a3075b42e9000f0b5293d7347d5e540a17755edb9d5b5","observation_id":"a7432697-617b-4d76-9b68-62afca620f25","resolution":{"observed_at":"2026-08-06T22:22:37.974207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.01818","last_updated":"2025-05-11T17:45:02Z","snapshot_observed_at":"2026-08-08T00:45:04.947316Z","submitted_at":"2024-12-02T18:57:40Z","title":"Beyond Text-Visual Attention: Exploiting Visual Cues for Effective Token Pruning in VLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.01818","snapshot_observed_at":"2026-08-06T22:22:38.007759Z","title":"[cls] attention is all you need for training- free visual token pruning: Make vlm inference faster","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:38.007759Z"},"links":{"cited_paper":"/paper/2412.01818","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:da369f0af794d307c2450061aa7ba1305fd884e7a04f362b8a97b656e5db72f3","observation_id":"9769bad8-5242-44f0-8796-3fb811209059","resolution":{"observed_at":"2026-08-06T22:22:38.007759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:38.460845Z","title":"Sparsevlm: Visual token sparsification for efficient vision- language model inference","venue":null,"work_id":"d6f96e8c-f04c-480a-932c-7ee1152e0a65","year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:38.056203Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:c210df20f5edae91f986702037a69423e1bf5d69f1c5389cea8f01f6474e8e42","observation_id":"8406bbc7-9c0c-4615-b864-0bf5b81d436a","resolution":{"observed_at":"2026-08-06T22:22:38.533842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T07:59:16.819249Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning"},"reference_resolution":{"displayed":33,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":18,"verified_exact":1,"verified_fuzzy":14},"total_outbound_references":33},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 33 of 33 outbound references and 4 inbound Pith citation observations for arXiv:2506.21873."}