{"as_of":"2026-08-14T15:16:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:190cf62acd0b463a7b7d362b2f753e3be7ec486d60b2f69b3cdbc5f023f088e5","coverage":[{"denominator":40,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T23:01:29.963011Z","state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:29:38.951111Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T17:29:40.174642Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"cited_work":{"arxiv_id":"2506.20066","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.20066","snapshot_observed_at":"2026-08-06T17:29:40.174642Z","title":"ToSA: Token Merging with Spatial Awareness","venue":"cs.CV","work_id":"fdd9354b-312d-4f55-ac77-d3bd25e98c6c","year":2025},"citing_paper":{"arxiv_id":"2507.10778","last_updated":"2025-08-14T03:48:03Z","snapshot_observed_at":"2026-08-11T16:37:34.930546Z","submitted_at":"2025-07-14T20:05:55Z","title":"Warehouse Spatial Question Answering with LLM Agent","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T17:29:38.951111Z"},"links":{"cited_paper":"/paper/2506.20066","citing_paper":"/paper/2507.10778"},"observation_digest":"sha256:d574c64b5bc4db4d13380bc7ed35c32a2f296e22252e0976285f8cd05589f644","observation_id":"91eac91c-5b43-4613-918e-ed4b68643fa5","resolution":{"observed_at":"2026-08-06T17:29:40.213090Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.20066","snapshot_observed_at":"2026-07-31T23:59:16.247451Z","title":"Tosa: To- ken merging with spatial awareness.arXiv preprint arXiv:2506.20066, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23265","last_updated":"2026-08-03T14:35:14Z","snapshot_observed_at":"2026-08-06T23:32:32.715200Z","submitted_at":"2026-07-25T16:05:22Z","title":"WaveZip: Wavelet-Driven Space-Time Decoupling for Video Token Condensation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-31T23:59:16.247451Z"},"links":{"cited_paper":"/paper/2506.20066","citing_paper":"/paper/2607.23265"},"observation_digest":"sha256:c281db95c865b6b44479649a9504c9492ba9b927d3094f477d06893a7e71219f","observation_id":"1ff61093-2310-4b77-bb4b-06a867db39d9","resolution":{"observed_at":"2026-07-31T23:59:16.247451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.20066","snapshot_observed_at":"2026-08-04T04:01:48.518353Z","title":"Tosa: To- ken merging with spatial awareness.arXiv preprint arXiv:2506.20066, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23265","last_updated":"2026-08-03T14:35:14Z","snapshot_observed_at":"2026-08-06T23:32:32.715200Z","submitted_at":"2026-07-25T16:05:22Z","title":"WaveZip: Wavelet-Driven Space-Time Decoupling for Video Token Condensation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T04:01:48.518353Z"},"links":{"cited_paper":"/paper/2506.20066","citing_paper":"/paper/2607.23265"},"observation_digest":"sha256:7be1ceddbebe8b4961bf12b47bbd0d4e4910a118746531eb27544b94a1f5d1ff","observation_id":"974bb5d5-894c-4430-9d4f-4be4e26c6499","resolution":{"observed_at":"2026-08-04T04:01:48.518353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.20066/citation-record","integrity":"/paper/2506.20066/integrity","json":"/paper/2506.20066/citation-record.json","paper":"/paper/2506.20066"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.308984Z","title":"Dinov2: Learning robust visual features without supervision,","venue":null,"work_id":"363445ab-b365-447a-926e-65449f6b577c","year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.840335Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:88f8de80473d0386588371d8a88ed621bd987b846ef7e22c436e0e0859008f0a","observation_id":"1e43f906-9452-4009-90bd-a78c62ddc673","resolution":{"observed_at":"2026-08-06T23:01:30.312138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:29.843839Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.843839Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:93e031cbc86efdb9ed0d31e07ffd9aa5be81b7e2418e31e307d00abc9684c26d","observation_id":"31346f8b-c886-440a-8a6c-0d45e01fb506","resolution":{"observed_at":"2026-08-06T23:01:29.843839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.293823Z","title":"Sigmoid loss for language image pre-training,","venue":null,"work_id":"f6aa8d09-36f2-4f0c-9426-c69de2b51d2f","year":2023},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.848181Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:c485d8a1898c9824248a45c85bec19375f69311921b46b138331ac9a34b9bc94","observation_id":"47d741a3-4730-4439-8a42-c0a638032c5f","resolution":{"observed_at":"2026-08-06T23:01:30.296994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T14:19:26.598265Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-06T23:01:29.852111Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.852111Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:098a47ced00b41ff8a0f1f7352a758131150e283ff9b6360df31c23f85dba481","observation_id":"a04e200e-2497-4f6e-b485-3f1c3baefbb5","resolution":{"observed_at":"2026-08-06T23:01:29.852111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:29.855346Z","title":"Visual instruction tuning,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.855346Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:57d11c0695b2cb15bd8eca9c7c2c931d4ce91d2179116fd288cd46623593e813","observation_id":"72db5e3e-9c2b-4405-babd-6f01a8e834c0","resolution":{"observed_at":"2026-08-06T23:01:29.855346Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-06T23:01:29.858437Z","title":"Llava-onevision: Easy visual task transfer,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.858437Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:142299c15555a0838790b5ffe10a86a8d750a2cde2e15555446ef06158665bcd","observation_id":"a940bb97-d4fd-4cf9-8501-0c57c3225c6c","resolution":{"observed_at":"2026-08-06T23:01:29.858437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.278664Z","title":"Efficientvit: Memory efficient vision transformer with cascaded group attention,","venue":null,"work_id":"0cbb68ca-a45a-48a7-8348-614fbb8ffc74","year":2023},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.861924Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:43758d9643f0f01248a9175463bfb7cb114920714d8a41d1dc29a4edff120220","observation_id":"e74f5825-bd81-4f70-a135-506546da5152","resolution":{"observed_at":"2026-08-06T23:01:30.282019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.269148Z","title":"Dynamicvit: Efficient vision transformers with dynamic token sparsification,","venue":null,"work_id":"1844dee2-490a-459f-ac35-780ee48c14e3","year":2021},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.864786Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:3d25527f71e36f72a0aab4fdc566f7b4a43ddff0a1aef2d15504894aac5b8cce","observation_id":"9bd3f93e-bb79-45c5-a42a-7ae4eccb9a80","resolution":{"observed_at":"2026-08-06T23:01:30.272427Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.259600Z","title":"A-vit: Adaptive tokens for efficient vision transformer,","venue":null,"work_id":"7f4e4c97-900b-4390-873e-4b68d5c3e199","year":2022},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.867828Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:9a9c6b4bb45ff51cf31da297057c8fa80188d79409a33ff51526b7a3309c3c62","observation_id":"04f5c40e-20c1-44a1-bbde-7afa92299075","resolution":{"observed_at":"2026-08-06T23:01:30.262824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.01583","last_updated":"2025-05-02T21:00:17Z","snapshot_observed_at":"2026-08-07T15:56:56.874332Z","submitted_at":"2025-05-02T21:00:17Z","title":"TEMPURA: Temporal Event Masked Prediction and Understanding for Reasoning in Action","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.01583","snapshot_observed_at":"2026-08-06T23:01:29.871387Z","title":"Tempura: Temporal event masked prediction and understanding for reasoning in action,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.871387Z"},"links":{"cited_paper":"/paper/2505.01583","citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:968475057849a0d5e7d6f9a116786fadf618f0d73957617588303e132cb2fc7d","observation_id":"8f6813e0-f0c8-4423-bb16-19a4b436e0ba","resolution":{"observed_at":"2026-08-06T23:01:29.871387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.250575Z","title":"Token pooling in vision transformers for image classification,","venue":null,"work_id":"5d1d2fb1-33b2-423a-ac70-b599b7468a1e","year":2023},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.874526Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:8c4bc7af9c2b30d825bcbc75c546a63c8a51fcc02448437769056b32a03a284d","observation_id":"b4463c08-a9f5-423f-84a5-02b9c02edb06","resolution":{"observed_at":"2026-08-06T23:01:30.253657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.241099Z","title":"Zero-shot 3d question answering via voxel-based dynamic token compression,","venue":null,"work_id":"6b28de11-1cb7-43dc-92c2-a8055d4c803d","year":2025},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.877445Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:d1db0d5b02225f0460719854a42c4cd1ef31e8845f12b29127483b49e9e3b6f2","observation_id":"e64efe56-fc1a-4d30-bb07-24fd401a13f5","resolution":{"observed_at":"2026-08-06T23:01:30.244426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.231676Z","title":"Token merging: Your ViT but faster,","venue":null,"work_id":"9d52c637-f66a-4edb-ad40-feb13060e82c","year":2023},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.880629Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:7a51ef6ed4522e20258f152a140d918f93bc1fc403fdc1545de5f22149d078ef","observation_id":"7d9d162f-cc20-4684-bca7-c8051db30524","resolution":{"observed_at":"2026-08-06T23:01:30.235279Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06727","last_updated":"2022-12-13T16:55:12Z","snapshot_observed_at":"2026-08-13T13:22:40.427545Z","submitted_at":"2022-12-13T16:55:12Z","title":"What do Vision Transformers Learn? A Visual Exploration","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.06727","snapshot_observed_at":"2026-08-06T23:01:29.883551Z","title":"What do vision transformers learn? a visual exploration,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.883551Z"},"links":{"cited_paper":"/paper/2212.06727","citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:cc33a03192ff73add0d82dea7188ae3a8a45e8778426014469960a2ef8d91df4","observation_id":"71ebff2d-17a0-4d8a-b8e2-2956bc460053","resolution":{"observed_at":"2026-08-06T23:01:29.883551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-13T16:11:30.018425Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T23:01:29.887022Z","title":"Spatialbot: Precise spatial understanding with vision lan- guage models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.887022Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:c6821a7ba108e9b134c2be5b04c7b1d478979943e20c8169367ede1ab8417a7e","observation_id":"ae73add2-f905-443d-b7aa-abbab6e06216","resolution":{"observed_at":"2026-08-06T23:01:29.887022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:29.890588Z","title":"Vqa: Visual question answering,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.890588Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:d6964c791cbf90ea0c704a76fb8e6d407e49dda93a58691f1e6feb12b7229a3f","observation_id":"adb71e1e-120d-46fa-8bf5-a8ff8a7e430e","resolution":{"observed_at":"2026-08-06T23:01:29.890588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.217096Z","title":"Gqa: A new dataset for real- world visual reasoning and compositional question answering,","venue":null,"work_id":"3ed2af22-f10a-4675-ac0b-76bfbd5a29ca","year":2019},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.894119Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:42f608a1b834def8444654df78ba5915cae4a68768f23563d144a1a24011d198","observation_id":"5817ad76-faa0-4ab9-a8ca-94e0cb8a63ae","resolution":{"observed_at":"2026-08-06T23:01:30.220159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.208495Z","title":"Openeqa: Embodied question answering in the era of foundation models,","venue":null,"work_id":"35cf8093-9d23-45ac-81ea-99da9e41e1cc","year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.896953Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:53b9927c1376aece26a8b41372152d9ca3fd4c1a04d556ed90e3f1c327de53b6","observation_id":"6bab8da7-a90f-4414-be8d-514a9a4a2bf9","resolution":{"observed_at":"2026-08-06T23:01:30.211596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.199084Z","title":"Sp-vit: Learning 2d spatial priors for vision transformers,","venue":null,"work_id":"39f18f5b-6032-4d5a-8893-731c2fefa513","year":2022},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.899813Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:8aa1e769579b0a89ba125c71e07a601d21ddd65a855ba00bb4c92affa796f5db","observation_id":"aa4213b9-0494-4f27-ba2b-1861cf231b88","resolution":{"observed_at":"2026-08-06T23:01:30.202142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.190147Z","title":"Evo-vit: Slow-fast token evolution for dynamic vision transformer,","venue":null,"work_id":"b66c8590-1aee-4bde-93ea-d51d77021462","year":2022},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.902805Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:37f129f1a31098acf842283768f2ae152763812a2505ae6b72f057a12b1b2713","observation_id":"d9f1d06b-b422-4d58-aba9-6c12e1392931","resolution":{"observed_at":"2026-08-06T23:01:30.193380Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.180565Z","title":"Not all patches are what you need: Expediting vision transformers via token reorganizations,","venue":null,"work_id":"bf7cdc2c-8e38-4de2-b38f-4f1deb9c32fb","year":2022},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.905853Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:0df495aa4291c08d55a3b0b47efbbb0bef199b4012f6cd63b1122a67fd74c5ce","observation_id":"3f03106a-f2a8-43c6-a505-631d9b8d6fa9","resolution":{"observed_at":"2026-08-06T23:01:30.183813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01812","last_updated":"2024-02-05T09:21:28Z","snapshot_observed_at":"2026-08-13T05:58:41.984751Z","submitted_at":"2023-10-03T05:55:11Z","title":"PPT: Token Pruning and Pooling for Efficient Vision Transformers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01812","snapshot_observed_at":"2026-08-06T23:01:29.908663Z","title":"Ppt: Token prun- ing and pooling for efficient vision transformers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.908663Z"},"links":{"cited_paper":"/paper/2310.01812","citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:e352b49fed4d2779f478cd6403b4c1c68693d51253f235945125e5d06550757d","observation_id":"718ac7db-d307-450e-a7d2-bd35898f18f9","resolution":{"observed_at":"2026-08-06T23:01:29.908663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.171549Z","title":"An image is worth 1/2 tokens after layer 2: Plug-and-play inference ac- celeration for large vision-language models,","venue":null,"work_id":"d3a20ac3-66f4-4b5e-a90e-553d0cf123fc","year":2025},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.911840Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:d1d4b02d8e6cf15592770e212279cbef966dcd2169dbb7ef23be36c66a294ec8","observation_id":"d7efb4b4-3a43-4e95-a7fc-162937ae0998","resolution":{"observed_at":"2026-08-06T23:01:30.174741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.160640Z","title":"Sparsevlm: Visual token sparsification for efficient vision-language model inference,","venue":null,"work_id":"b42f34e9-1c62-4ec6-aac2-4d1e33c3d195","year":2025},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.914750Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:fbdb830f61e443246fc62321962789c1e9a2a439e886473436bb3456d11a247d","observation_id":"e2a3183e-3954-428f-8d4d-1c3037f0678d","resolution":{"observed_at":"2026-08-06T23:01:30.164160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.151005Z","title":"Spatialvlm: Endowing vision-language models with spatial reasoning capabilities,","venue":null,"work_id":"83416c55-b362-4430-aabb-d8f360d00e16","year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.917634Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:11df9aa4fe8efe2fc4d1a5fc0257d5e013e2ed9b675f8fa2b57d19fb8366d4ec","observation_id":"69f882be-bbcd-4864-9259-482ec2173ddb","resolution":{"observed_at":"2026-08-06T23:01:30.154332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.141833Z","title":"Spatialrgpt: Grounded spatial reasoning in vision-language models,","venue":null,"work_id":"932bce33-ffc6-480a-a2be-3ce106df06cb","year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.920675Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:aef1d6abf3db6e608b77ed02be6b7be2a1124c4279ff7386ae626376d654c173","observation_id":"f2b6a679-237c-449a-ba4d-f6d133ceca6a","resolution":{"observed_at":"2026-08-06T23:01:30.145170Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:29.923678Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.923678Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:e6ca4441a7d89fed470ff793e329717b9be38b1962517193e526cf9302bfe9e1","observation_id":"af21c5cc-af7a-428d-8816-33de85b0785c","resolution":{"observed_at":"2026-08-06T23:01:29.923678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:29.926930Z","title":"Blip-2: Bootstrapping language- image pre-training with frozen image encoders and large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.926930Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:c8513b3238e68ded2e1580f44c5060b30ce75c8f908aa45c0afb73c0695b3783","observation_id":"ea6ef157-f9ed-4beb-8122-dba2c9bdb969","resolution":{"observed_at":"2026-08-06T23:01:29.926930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:29.929843Z","title":"Instructblip: Towards general-purpose vision- language models with instruction tuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.929843Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:319584bf6e986b8e8e1589910ea3965d01c766c233b710e2b227352e085ec0df","observation_id":"bd4975ca-7cd5-41c2-a9d6-060bf6e5d0b1","resolution":{"observed_at":"2026-08-06T23:01:29.929843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-06T23:01:29.932806Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.932806Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:19a5831ecfee74a9c441cb47ee5223e87e4dee18913bdec1c20e0a5c5487c9d3","observation_id":"242f7156-e487-47c6-8819-3b72cb02be43","resolution":{"observed_at":"2026-08-06T23:01:29.932806Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:29.936217Z","title":"Improved baselines with visual instruction tuning,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.936217Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:15f2d9fc0941ee1a1472f32de7e1eab747733c374425bae0d97555afb4ecbbb4","observation_id":"a3e07ea0-fdbd-4e2b-8c53-1f464f6b9e09","resolution":{"observed_at":"2026-08-06T23:01:29.936217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:29.939078Z","title":"Llama-vid: An image is worth 2 tokens in large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.939078Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:6b4e6b5172659675050083c07762b9c8bea6ca2d45eb2a9e0f3adedea08dfc80","observation_id":"015df59e-fd2d-49de-b961-fa434bf38c49","resolution":{"observed_at":"2026-08-06T23:01:29.939078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.106492Z","title":"Vila: On pre-training for visual language models,","venue":null,"work_id":"43d7d81a-4985-4ec0-b100-82f4a5bbd587","year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.941985Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:9a656f3d087edc5d85feb8ddb2a11d12cca04787131aefad07571bcdd093f865","observation_id":"7563aa29-e8bf-48af-b677-46ce31a1056b","resolution":{"observed_at":"2026-08-06T23:01:30.109792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.096937Z","title":"Depth anything v2,","venue":null,"work_id":"bd055b87-0ada-41bb-b6a1-278779cec912","year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.944983Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:ab3bf44b433dc76b7fc3c3abfb2315b43203b50865bed7adc827bf5505e69c80","observation_id":"8f2e361f-155a-49e1-8770-0abe81f21a22","resolution":{"observed_at":"2026-08-06T23:01:30.100457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03051","last_updated":"2025-04-09T06:24:14Z","snapshot_observed_at":"2026-08-14T03:26:31.108831Z","submitted_at":"2024-10-04T00:13:54Z","title":"AuroraCap: Efficient, Performant Video Detailed Captioning and a New Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03051","snapshot_observed_at":"2026-08-06T23:01:29.947783Z","title":"Auroracap: Efficient, performant video detailed captioning and a new benchmark,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.947783Z"},"links":{"cited_paper":"/paper/2410.03051","citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:6aefb43d262b3c3c50636d68caaf38c29d840604cc4ac35cf825601c87ddb69e","observation_id":"3bf9bb16-7765-40f5-b839-9587f6f7bfcb","resolution":{"observed_at":"2026-08-06T23:01:29.947783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.088195Z","title":"Longvlm: Efficient long video understanding via large language models,","venue":null,"work_id":"7375da6c-c558-454b-b6af-c52fcb82072a","year":2025},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.951218Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:33b7aaf6a450bfac2188bcfb3a7491e50113f74e7fa0f0b31a239b10da7fb35d","observation_id":"bcb28fe4-92e5-4dba-ab9a-101a3e869823","resolution":{"observed_at":"2026-08-06T23:01:30.091338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.079188Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models,","venue":null,"work_id":"6a92eb39-9e24-4384-b069-1d165f7e6641","year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.954230Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:671c2f0816761a79e72f393d37ab589a4a7e796b7398a01b6c0312955115c419","observation_id":"8f28d1d8-cdee-4d34-8e7b-83e03edac955","resolution":{"observed_at":"2026-08-06T23:01:30.082347Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.069341Z","title":"Video-llama: An instruction-tuned audio-visual language model for video understanding,","venue":null,"work_id":"3b178e5a-1db9-4fc0-8a49-ce81c092b8cb","year":2023},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.957084Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:3b041f127f1e596b85c41eb0063b0ff634759eff41708a91ceb734710be5c917","observation_id":"2b74cc36-57e4-4d1d-99bb-022d8b87d674","resolution":{"observed_at":"2026-08-06T23:01:30.072638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-06T23:01:29.959967Z","title":"Videollama 2: Advancing spatial- temporal modeling and audio understanding in video-llms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.959967Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:515262e9df51682a9b14ddf92600edb9e3769154cadf35f34c5895d68f9da48b","observation_id":"fab74acb-a524-4e80-a5b2-bc01b79f2c1e","resolution":{"observed_at":"2026-08-06T23:01:29.959967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:01:30.057978Z","title":"Chat-univi: Unified visual representation empowers large language models with image and video understanding,","venue":null,"work_id":"e5efb98d-355f-4e02-ace6-00b84bbfaa0e","year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.963011Z"},"links":{"citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:7589afa6d63ff0e610b81f8f73fff2ac20946f5c3e71a8803e034fa821ae7e3d","observation_id":"f94ba385-b0bb-4504-a4ab-9870b150d347","resolution":{"observed_at":"2026-08-06T23:01:30.063216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness"},"reference_resolution":{"displayed":40,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":0,"verified_fuzzy":23},"total_outbound_references":40},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 40 of 40 outbound references and 3 inbound Pith citation observations for arXiv:2506.20066."}