{"as_of":"2026-08-20T10:58:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ce1611bcdb923f2ecc4cf092d025da7ccf6a261336a3e4bb4effc3756c5f045b","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T17:34:25.833722Z","state":"measured"},{"denominator":41,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":41,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T00:49:13.291897Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T21:08:58.805621Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"cited_work":{"arxiv_id":"2508.09032","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.09032","snapshot_observed_at":"2026-07-03T21:08:58.805621Z","title":"Spatial traces: Enhancing vla models with spatial-temporal understanding","venue":null,"work_id":"42c8bfd9-10b5-40d7-bc46-cf63683635f9","year":2025},"citing_paper":{"arxiv_id":"2508.13073","last_updated":"2025-09-01T08:10:01Z","snapshot_observed_at":"2026-08-07T15:40:31.068428Z","submitted_at":"2025-08-18T16:45:48Z","title":"Large VLM-based Vision-Language-Action Models for Robotic Manipulation: A Survey","version":2},"reference_index":110,"source":"pdf_text","source_observed_at":"2026-05-17T20:28:15.818016Z"},"links":{"cited_paper":"/paper/2508.09032","citing_paper":"/paper/2508.13073"},"observation_digest":"sha256:18d46ab8d345888509a4894f92eb97c129cda730a857e9d3bc44ba958b0fabd6","observation_id":"2a29d3a0-d9c2-4eea-a9e5-33468fe17bd4","resolution":{"observed_at":"2026-05-17T20:28:16.167232Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"cited_work":{"arxiv_id":"2508.09032","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.09032","snapshot_observed_at":"2026-07-03T21:08:58.805621Z","title":"Spatial traces: Enhancing vla models with spatial-temporal understanding","venue":null,"work_id":"42c8bfd9-10b5-40d7-bc46-cf63683635f9","year":2025},"citing_paper":{"arxiv_id":"2602.20231","last_updated":"2026-04-09T04:26:01Z","snapshot_observed_at":"2026-08-15T02:14:17.838458Z","submitted_at":"2026-02-23T18:41:41Z","title":"UniLACT: Depth-Aware RGB Latent Action Learning for Vision-Language-Action Models","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-15T20:18:31.988002Z"},"links":{"cited_paper":"/paper/2508.09032","citing_paper":"/paper/2602.20231"},"observation_digest":"sha256:368a0cbb7a45891b55c5314d2bd9b37d006931bceab24993dccce9ca1262486d","observation_id":"b5664072-da6e-481d-a15f-a72a0b2156b5","resolution":{"observed_at":"2026-05-15T20:20:17.611287Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"cited_work":{"arxiv_id":"2508.09032","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.09032","snapshot_observed_at":"2026-07-03T21:08:58.805621Z","title":"Spatial traces: Enhancing vla models with spatial-temporal understanding","venue":null,"work_id":"42c8bfd9-10b5-40d7-bc46-cf63683635f9","year":2025},"citing_paper":{"arxiv_id":"2606.17598","last_updated":"2026-08-12T12:13:59Z","snapshot_observed_at":"2026-08-17T12:22:24.569258Z","submitted_at":"2026-06-16T07:04:13Z","title":"MuseVLA: An Adaptive Multimodal Sensing Vision-Language-Action Model for Robotic Manipulation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T00:49:13.291897Z"},"links":{"cited_paper":"/paper/2508.09032","citing_paper":"/paper/2606.17598"},"observation_digest":"sha256:e397e054f6d2bf19e5a84b53984cb775a3a05824446266407bf3a40873552e3b","observation_id":"6dd17db8-49e1-424c-aa14-ecd4a8a41852","resolution":{"observed_at":"2026-07-03T21:08:58.807388Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2508.09032/citation-record","integrity":"/paper/2508.09032/integrity","json":"/paper/2508.09032/citation-record.json","paper":"/paper/2508.09032"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:25.542080Z","title":"Inner monologue: Embodied reasoning through planning with language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.542080Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:3f2b37367add9c972fdcbe391e68443e6a05919c0ec639aa1efa096c04b67c4f","observation_id":"4c79802e-87d9-49c9-918e-f28944f26181","resolution":{"observed_at":"2026-08-15T17:34:25.542080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:25.547596Z","title":"Application of pretrained large language models in embodied artificial intelligence,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.547596Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:f213d36aff03830d0800c6be362bb95d6a4baa5e0e6ca4e1481f9d0c4c3f0554","observation_id":"e50cfe5e-3f45-4182-9d51-6b138e65db61","resolution":{"observed_at":"2026-08-15T17:34:25.547596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:25.552715Z","title":"Evaluation of pretrained large language models in embodied planning tasks,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.552715Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:1d188096c16bdaf6fb6190494fa1a51ec5d01dd382a4a3a65b53410d93a873e8","observation_id":"7df9b5f0-a8af-4a57-a956-13f95bc731e5","resolution":{"observed_at":"2026-08-15T17:34:25.552715Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.444884Z","title":"Palm- e: An embodied multimodal language model,","venue":null,"work_id":"cecb0cff-d3fb-4602-bd92-4b09e8ecfc7e","year":2023},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.557331Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:2dcea243dd40568b7307710cc94d1c1a77f45e194571588164d304bd4fae0541","observation_id":"7a88b2c3-33d7-4794-8f4b-6eaac602a5fe","resolution":{"observed_at":"2026-08-15T17:34:26.449912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.430121Z","title":"Common sense plan verification with large language models,","venue":null,"work_id":"b70eb6ba-0f8e-4520-ab7e-f2be761e8ce6","year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.675049Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:796b61981ac584c047b414a771b72287a40098411718883699ff359e8ff2ab18","observation_id":"ba33e3a2-62e4-43f0-a00f-beda6566fed4","resolution":{"observed_at":"2026-08-15T17:34:26.434694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.05118","last_updated":"2025-07-07T15:31:36Z","snapshot_observed_at":"2026-08-20T08:02:59.240791Z","submitted_at":"2025-07-07T15:31:36Z","title":"VerifyLLM: LLM-Based Pre-Execution Task Plan Verification for Robots","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.05118","snapshot_observed_at":"2026-08-15T17:34:25.680454Z","title":"Verifyllm: Llm- based pre-execution task plan verification for robots,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.680454Z"},"links":{"cited_paper":"/paper/2507.05118","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:391c65d2cfe0146ac8439701e066b410eeaf5af040824ced2b362a7cdec2c1ed","observation_id":"362e8a88-eb42-440c-bc1b-c59613465990","resolution":{"observed_at":"2026-08-15T17:34:25.680454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.414342Z","title":"Llm-planner: Few-shot grounded planning for embodied agents with large language models,","venue":null,"work_id":"3844c028-eba2-4e90-a7eb-77bafce9820f","year":2023},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.685823Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:a899c144609c6920d9ba2fc753f7066879112709181c427978ebb24f301d979b","observation_id":"27cd7e8c-c175-4aff-9c68-87813bca387d","resolution":{"observed_at":"2026-08-15T17:34:26.419539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17842","last_updated":"2023-12-24T03:48:40Z","snapshot_observed_at":"2026-08-20T00:31:41.370968Z","submitted_at":"2023-11-29T17:46:25Z","title":"Look Before You Leap: Unveiling the Power of GPT-4V in Robotic Vision-Language Planning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17842","snapshot_observed_at":"2026-08-15T17:34:25.690219Z","title":"Look before you leap: Unveiling the power of gpt-4v in robotic vision-language planning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.690219Z"},"links":{"cited_paper":"/paper/2311.17842","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:4eb1e019af9f9a495d3ffb0ebffd491457f04eca1f6b09c32225bb600f1747d4","observation_id":"f385eb77-eff2-455d-a1e6-ca83ad8f847a","resolution":{"observed_at":"2026-08-15T17:34:25.690219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:25.694982Z","title":"Lookplangraph: Embodied instruction following method with VLM graph augmentation,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.694982Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:afd2b9645ccf72ba5e246d61ba1b3b464ec8b26058a30bbfb447d75d7d458663","observation_id":"f10d1671-62ba-4553-9245-7642b2582cb3","resolution":{"observed_at":"2026-08-15T17:34:25.694982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.388854Z","title":"Lera: Replanning with visual feedback in instruction following,","venue":null,"work_id":"70b68c05-14c1-4d70-9a0e-0165d3460a73","year":2025},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.699553Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:2916d77747660db26999c99598ef461b69e715dc5a219445c2034fb1349dd8d1","observation_id":"c5f25143-2221-4230-a4cb-d9ca78cdc3c4","resolution":{"observed_at":"2026-08-15T17:34:26.393805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03094","last_updated":"2023-05-28T07:32:38Z","snapshot_observed_at":"2026-08-16T16:26:20.587132Z","submitted_at":"2022-10-06T17:50:11Z","title":"VIMA: General Robot Manipulation with Multimodal Prompts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03094","snapshot_observed_at":"2026-08-15T17:34:25.703990Z","title":"Vima: General robot manip- ulation with multimodal prompts,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.703990Z"},"links":{"cited_paper":"/paper/2210.03094","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:e8fce37430eb4f6badeaa859fa53f2dab1ec88ef8582da362c62bb3ad68efd1f","observation_id":"661dc391-4b9b-491e-9677-72f612a0bf8f","resolution":{"observed_at":"2026-08-15T17:34:25.703990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.373424Z","title":"Fine-tuning multi- modal transformer models for generating actions in virtual and real environments,","venue":null,"work_id":"05c146f7-17e7-4399-9b47-6fd3aa011f3e","year":null},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.708901Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:19e053299d5d38e6ffff88c30e696238bdf77eda17d24d39fe293b1d7dce66c9","observation_id":"519e80d1-1c3c-4ed9-b04b-b5a2f2d0f326","resolution":{"observed_at":"2026-08-15T17:34:26.378478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:25.713763Z","title":"Octo: An open-source generalist robot policy,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.713763Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:a1e3cc4e9583251fc6f093e74de2b3f3a7344132d8812c1229cd5c7661c2d4d0","observation_id":"445cdcaa-2446-4380-a9a8-32ba03dbba72","resolution":{"observed_at":"2026-08-15T17:34:25.713763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09246","last_updated":"2024-09-05T19:46:34Z","snapshot_observed_at":"2026-08-16T21:53:14.144225Z","submitted_at":"2024-06-13T15:46:55Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09246","snapshot_observed_at":"2026-08-15T17:34:25.718240Z","title":"Openvla: An open-source vision-language-action model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.718240Z"},"links":{"cited_paper":"/paper/2406.09246","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:56d57179575384d764c642ff087b1ce84fa8373fd2aa5f294ab0db54c03cfd9a","observation_id":"b5dfc1e8-a6c5-46f0-be24-8517420a92e7","resolution":{"observed_at":"2026-08-15T17:34:25.718240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.19650","last_updated":"2024-11-29T12:06:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-29T12:06:03Z","title":"CogACT: A Foundational Vision-Language-Action Model for Synergizing Cognition and Action in Robotic Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.19650","snapshot_observed_at":"2026-08-15T17:34:25.722997Z","title":"Cogact: A foundational vision- language-action model for synergizing cognition and action in robotic manipulation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.722997Z"},"links":{"cited_paper":"/paper/2411.19650","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:d81688883becb0569dcdf25b72415e75aada40766b1ee3b3d61b0ab866459d7a","observation_id":"1b91a08a-166a-43c7-9bbc-cb7f3ae0251e","resolution":{"observed_at":"2026-08-15T17:34:25.722997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15830","last_updated":"2025-05-19T02:40:18Z","snapshot_observed_at":"2026-08-20T05:57:34.826209Z","submitted_at":"2025-01-27T07:34:33Z","title":"SpatialVLA: Exploring Spatial Representations for Visual-Language-Action Model","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15830","snapshot_observed_at":"2026-08-15T17:34:25.727844Z","title":"Spatialvla: Exploring spatial representations for visual-language-action model,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.727844Z"},"links":{"cited_paper":"/paper/2501.15830","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:e603430bf3ea04e079f292c9c040a37f602bf593d7dfcc08cba53cb573caf59a","observation_id":"7ff86df5-5a33-4c63-a3df-c7dddc97c685","resolution":{"observed_at":"2026-08-15T17:34:25.727844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.347668Z","title":"Robovqa: Multimodal long-horizon reasoning for robotics,","venue":null,"work_id":"445cfc22-ae73-459e-bf85-bda54051301f","year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.732838Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:a4e17f19194bf508f7e4182cb8ef5de10e1a6c6b9c8b799caba50f7b3155802e","observation_id":"d1e8a458-9d5f-432a-bcd1-7596f83da34d","resolution":{"observed_at":"2026-08-15T17:34:26.352672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.331253Z","title":"Mastering long-context multi-task reasoning with transformers and recurrent memory,","venue":null,"work_id":"c42ab9fc-c205-4dbb-b10c-47af8d64860d","year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.737497Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:cd31db652ce109bd10b99394f1193d0cb1d96532221ab505160e2cd65920814c","observation_id":"575497c8-4f22-48ca-94b1-1982e83188bc","resolution":{"observed_at":"2026-08-15T17:34:26.336917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.315311Z","title":"Babilong: Testing the limits of llms with long con- text reasoning-in-a-haystack,","venue":null,"work_id":"10f9195d-3c0b-4834-8133-cb3377fe1d67","year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.742016Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:7ab712f94a4a8265651287f51299048c5cecd5e4910141687f46c5d7bc70bc98","observation_id":"afab4d89-87c2-48d7-941c-16fd576c3400","resolution":{"observed_at":"2026-08-15T17:34:26.320255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.01977","last_updated":"2023-11-06T05:53:08Z","snapshot_observed_at":"2026-08-16T14:46:11.532367Z","submitted_at":"2023-11-03T15:31:51Z","title":"RT-Trajectory: Robotic Task Generalization via Hindsight Trajectory Sketches","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.01977","snapshot_observed_at":"2026-08-15T17:34:25.746610Z","title":"Rt-trajectory: Robotic task generalization via hindsight trajectory sketches,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.746610Z"},"links":{"cited_paper":"/paper/2311.01977","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:039e8a96f44ce4154b2fe7c35c5d716cd3fef26aec46eb430cff9c6977b7641a","observation_id":"862d4eed-8104-40b9-bfd3-6ee28530e885","resolution":{"observed_at":"2026-08-15T17:34:25.746610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10345","last_updated":"2025-06-05T21:26:08Z","snapshot_observed_at":"2026-08-12T18:17:40.083518Z","submitted_at":"2024-12-13T18:40:51Z","title":"TraceVLA: Visual Trace Prompting Enhances Spatial-Temporal Awareness for Generalist Robotic Policies","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10345","snapshot_observed_at":"2026-08-15T17:34:25.751730Z","title":"Tracevla: Visual trace prompting enhances spatial-temporal awareness for generalist robotic policies,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.751730Z"},"links":{"cited_paper":"/paper/2412.10345","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:b2c6786b2e05b1064acd93e1b16582c22ba35f4f1f1ffe3ec04e7c16ebb74211","observation_id":"f01f22ae-6a1d-4d0d-bdd6-6350199ca179","resolution":{"observed_at":"2026-08-15T17:34:25.751730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02704","last_updated":"2024-11-05T01:02:51Z","snapshot_observed_at":"2026-08-16T13:02:56.960449Z","submitted_at":"2024-11-05T01:02:51Z","title":"RT-Affordance: Affordances are Versatile Intermediate Representations for Robot Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02704","snapshot_observed_at":"2026-08-15T17:34:25.756414Z","title":"Rt-affordance: Affordances are versatile intermediate representations for robot manipulation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.756414Z"},"links":{"cited_paper":"/paper/2411.02704","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:12ba440e13332f3509e1ea4073ff79fda121409f490e51aac6ab6736510db873","observation_id":"8f8fadbd-8a5a-42f5-833a-911185d0387d","resolution":{"observed_at":"2026-08-15T17:34:25.756414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08864","last_updated":"2025-05-14T15:22:36Z","snapshot_observed_at":"2026-08-13T13:59:48.091257Z","submitted_at":"2023-10-13T05:20:40Z","title":"Open X-Embodiment: Robotic Learning Datasets and RT-X Models","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08864","snapshot_observed_at":"2026-08-15T17:34:25.761568Z","title":"Open X-Embodiment: Robotic learning datasets and RT-X models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.761568Z"},"links":{"cited_paper":"/paper/2310.08864","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:4a6207c0adf5b955ea84de60478eb5282424d89238c479c33c4a941a37599a93","observation_id":"7343542e-ba85-4d74-80f0-4352ebac6e93","resolution":{"observed_at":"2026-08-15T17:34:25.761568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:25.767070Z","title":"Sem: Enhancing spatial understanding for robust robot manipulation,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.767070Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:6706bb8fc17f36b93c96f3000a7936781c96b065a71a6a176a7a2d5e52edcb39","observation_id":"709475e8-3d5e-4c86-989c-a309ea588af4","resolution":{"observed_at":"2026-08-15T17:34:25.767070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15975","last_updated":"2023-08-31T15:29:44Z","snapshot_observed_at":"2026-08-16T15:04:38.131135Z","submitted_at":"2023-08-30T11:57:04Z","title":"RoboTAP: Tracking Arbitrary Points for Few-Shot Visual Imitation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.15975","snapshot_observed_at":"2026-08-15T17:34:25.771689Z","title":"Robotap: Tracking arbitrary points for few-shot visual imitation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.771689Z"},"links":{"cited_paper":"/paper/2308.15975","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:61cbbff547f644f07f5eca5b621213947651b8594fd63ac54773377e25e58d86","observation_id":"6dc7c7ce-30d6-4f84-8998-520fd08e5270","resolution":{"observed_at":"2026-08-15T17:34:25.771689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.08685","last_updated":"2024-12-26T19:50:42Z","snapshot_observed_at":"2026-08-16T15:23:53.063232Z","submitted_at":"2023-06-14T18:10:05Z","title":"World-to-Words: Grounded Open Vocabulary Acquisition through Fast Mapping in Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2306.08685","doi":null,"metadata_source":"pith","pith_arxiv_id":"2306.08685","snapshot_observed_at":"2026-08-15T17:34:25.967585Z","title":"World-to-Words: Grounded Open Vocabulary Acquisition through Fast Mapping in Vision-Language Models","venue":"cs.CL","work_id":"b5cb1fb5-0e3b-4eb7-ab1e-7158fd3d1e23","year":2023},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.777343Z"},"links":{"cited_paper":"/paper/2306.08685","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:bfabb59cb9910a1ea5683d80851979edc9ea6c10ccaf7b48490e84b766df33b0","observation_id":"f4c68434-6c73-4a5f-90b3-5b5f38204027","resolution":{"observed_at":"2026-08-15T17:34:25.974845Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.05941","last_updated":"2024-05-09T17:30:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-09T17:30:16Z","title":"Evaluating Real-World Robot Manipulation Policies in Simulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.05941","snapshot_observed_at":"2026-08-15T17:34:25.781988Z","title":"Cotracker v3: Joint tracking of multiple points via transformer,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.781988Z"},"links":{"cited_paper":"/paper/2405.05941","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:fca46d08e5e9f27ff6b0ee1e3782e26dfcdf324ba9e104b9e77f10afbed90980","observation_id":"446949bb-978c-49ef-b901-6fdf9ff1c1ab","resolution":{"observed_at":"2026-08-15T17:34:25.781988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13130","last_updated":"2025-02-18T18:55:21Z","snapshot_observed_at":"2026-08-16T12:57:13.112901Z","submitted_at":"2025-02-18T18:55:21Z","title":"Magma: A Foundation Model for Multimodal AI Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13130","snapshot_observed_at":"2026-08-15T17:34:25.786667Z","title":"Magma: A foundation model for multimodal ai agents,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.786667Z"},"links":{"cited_paper":"/paper/2502.13130","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:e88674e9cf053a026dbb4e1f42ba2ce65f8f8ead1845c2b5ac36fd546e7ba460","observation_id":"5073f28c-aad3-4f17-9c3a-72165fc0c8cc","resolution":{"observed_at":"2026-08-15T17:34:25.786667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.299739Z","title":"Sigmoid loss for language image pre-training,","venue":null,"work_id":"59e4034e-41ac-4b68-b4a4-0f2f87533d04","year":2023},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.791653Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:d79f712888e44dd34bbbc8cc4ae4414d86920d9ca29d14ae59a860cbce184137","observation_id":"259d631a-a390-4e41-84e1-78ce45374c78","resolution":{"observed_at":"2026-08-15T17:34:26.304612Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.283963Z","title":"Cotracker: It is better to track together,","venue":null,"work_id":"6a1f99cf-a3fc-419f-8cbd-0c66d31bb059","year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.796117Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:e51b0bd4dbb2ab560401ed3caeca21f0cd7b528938a0d4093eb90ab7fe4bc3ee","observation_id":"fb5df039-beb0-426a-8126-77188708c701","resolution":{"observed_at":"2026-08-15T17:34:26.288849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.12288","last_updated":"2023-02-23T19:13:10Z","snapshot_observed_at":"2026-07-06T14:55:15.719380Z","submitted_at":"2023-02-23T19:13:10Z","title":"ZoeDepth: Zero-shot Transfer by Combining Relative and Metric Depth","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.12288","snapshot_observed_at":"2026-08-15T17:34:25.800692Z","title":"Zoedepth: Zero-shot transfer by combining relative and metric depth,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.800692Z"},"links":{"cited_paper":"/paper/2302.12288","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:774741c2e014bcb2c6d83c785d5c2fba76c92d73710964a88bfebd75cb1834da","observation_id":"732633ba-e588-490f-b434-6b081c09498d","resolution":{"observed_at":"2026-08-15T17:34:25.800692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03555","last_updated":"2024-12-04T18:50:42Z","snapshot_observed_at":"2026-07-06T20:01:45.826971Z","submitted_at":"2024-12-04T18:50:42Z","title":"PaliGemma 2: A Family of Versatile VLMs for Transfer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.03555","snapshot_observed_at":"2026-08-15T17:34:25.805861Z","title":"Paligemma 2: A family of versatile vlms for transfer,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.805861Z"},"links":{"cited_paper":"/paper/2412.03555","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:c7b4e217b23290c06a75c7f412bd6722fb278640ac4b06c140cb7c52473289c0","observation_id":"ca905dde-d301-4d60-8301-9c66a6a2328d","resolution":{"observed_at":"2026-08-15T17:34:25.805861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.268913Z","title":"Evaluating real-world robot manipulation policies in simulation,","venue":null,"work_id":"f36e8957-115d-4e74-aaf3-187a8065f222","year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.810457Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:ca8215e1fc4591fb056dedd00407df7ff1bfb184d71676792a316afe663b21bd","observation_id":"3470ffc9-d1b7-4435-93d4-d206091ef020","resolution":{"observed_at":"2026-08-15T17:34:26.273758Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.12293","last_updated":"2025-01-18T02:57:38Z","snapshot_observed_at":"2026-08-19T01:31:41.184872Z","submitted_at":"2020-09-25T15:32:31Z","title":"robosuite: A Modular Simulation Framework and Benchmark for Robot Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.12293","snapshot_observed_at":"2026-08-15T17:34:25.814757Z","title":"robosuite: A modular simulation framework and benchmark for robot learning,","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.814757Z"},"links":{"cited_paper":"/paper/2009.12293","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:f651c22cd29e5398f605f94ef1d72e323c244cdcb3c3159d9c89ace90e1e18a1","observation_id":"20d6ac6e-e5b6-4aea-98cc-074334fc4432","resolution":{"observed_at":"2026-08-15T17:34:25.814757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14058","last_updated":"2026-02-13T02:05:15Z","snapshot_observed_at":"2026-08-17T09:33:59.536762Z","submitted_at":"2024-12-18T17:07:20Z","title":"What Matters in Building Vision-Language-Action Models for Generalist Robots","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14058","snapshot_observed_at":"2026-08-15T17:34:25.819442Z","title":"Towards generalist robot policies: What matters in building vision-language-action models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.819442Z"},"links":{"cited_paper":"/paper/2412.14058","citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:895dfa3772e4dd31d90c4d72a6e7a99b9605177b02fe30390372e97c5d555e89","observation_id":"adcc1023-9d2a-4357-bc9a-79654c1e0a0b","resolution":{"observed_at":"2026-08-15T17:34:25.819442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:26.252438Z","title":"Bridgedata v2: A dataset for robot learning at scale,","venue":null,"work_id":"9a2fd456-55e6-4f7c-9b55-7b5fdbe7aff7","year":2023},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.824184Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:ca82161700c8d7397a53321bfa0aa821c9f02fc0824e7390ad19cb36b35145d0","observation_id":"fce96eb1-110f-4e64-93c8-2faaad3f6794","resolution":{"observed_at":"2026-08-15T17:34:26.258080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:25.829203Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.829203Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:f2b645b7bc48bf8c17cca88a9506cad5a1f56ad354fa244c3e5dbdd0ed7e4604","observation_id":"8c452362-9e51-4aa5-ae8e-a55710f7966b","resolution":{"observed_at":"2026-08-15T17:34:25.829203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:34:25.833722Z","title":"Decoupled weight decay regularization,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T17:34:25.833722Z"},"links":{"citing_paper":"/paper/2508.09032"},"observation_digest":"sha256:1f5d93da81e1bb8e0c0287317319523fef04d7427dfebead9f04f62654dbdfb6","observation_id":"d31ee40e-ac83-4eb2-9b8f-51a197c954c8","resolution":{"observed_at":"2026-08-15T17:34:25.833722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2508.09032","last_updated":"2025-08-12T15:53:45Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-20T04:51:43.075806Z","submitted_at":"2025-08-12T15:53:45Z","title":"Spatial Traces: Enhancing VLA Models with Spatial-Temporal Understanding"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":25,"verified_exact":0,"verified_fuzzy":12},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 3 inbound Pith citation observations for arXiv:2508.09032."}