{"as_of":"2026-08-09T09:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:496ceeb49563c104169a5fb998343551dab25020e0dc6fabee86275a06b9a4c9","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":33,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T22:10:13.371215Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T04:09:35.261339Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2401.00025","last_updated":"2024-07-12T12:51:00Z","snapshot_observed_at":"2026-07-06T17:09:59.848387Z","submitted_at":"2023-12-28T23:34:43Z","title":"Any-point Trajectory Modeling for Policy Learning","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T23:32:50.916085Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2401.00025"},"observation_digest":"sha256:d40c78d490f2626e9954e014ad8c0a286e638b21b344fef5b649abc879feac29","observation_id":"3aeba27b-6d61-45e8-823f-61b7b3fc3d7e","resolution":{"observed_at":"2026-05-16T23:32:50.976352Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2401.03568","last_updated":"2024-01-25T21:20:27Z","snapshot_observed_at":"2026-07-30T06:39:40.880794Z","submitted_at":"2024-01-07T19:11:18Z","title":"Agent AI: Surveying the Horizons of Multimodal Interaction","version":2},"reference_index":130,"source":"arxiv_source","source_observed_at":"2026-05-18T14:25:58.876978Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2401.03568"},"observation_digest":"sha256:f8e93edc5a923c33d4910359c8a8c860d8fd07f699a2e07d63fa795b6a026ee8","observation_id":"8c4d3cbd-e360-410f-8c90-b25e27309042","resolution":{"observed_at":"2026-05-18T14:25:59.387306Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2411.04983","last_updated":"2025-02-01T02:40:49Z","snapshot_observed_at":"2026-07-06T19:46:54.707852Z","submitted_at":"2024-11-07T18:54:37Z","title":"DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-17T16:06:09.448517Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2411.04983"},"observation_digest":"sha256:3b0e3f2c3a23fd5130456387c11353554df5a4d430016594290c5b22c08e29b6","observation_id":"bb09e5b1-9a2a-4aa0-a4e6-37be6eeaa1ea","resolution":{"observed_at":"2026-05-17T16:06:09.770224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-08-07T22:10:13.371215Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.09268","last_updated":"2025-02-14T01:51:57Z","snapshot_observed_at":"2026-08-08T03:06:57.057853Z","submitted_at":"2025-02-13T12:29:50Z","title":"GEVRM: Goal-Expressive Video Generation Model For Robust Visual Manipulation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T22:10:13.371215Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2502.09268"},"observation_digest":"sha256:b05ca382b30616591c538bf31a989ad714cb5cec109c70dc65f91bd44e84fa85","observation_id":"b85776dd-69b6-4ec7-b660-b452a7b2d027","resolution":{"observed_at":"2026-08-07T22:10:13.371215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-08-07T20:08:49.465586Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.09923","last_updated":"2025-02-14T05:23:56Z","snapshot_observed_at":"2026-08-08T05:52:10.090372Z","submitted_at":"2025-02-14T05:23:56Z","title":"Self-Consistent Model-based Adaptation for Visual Reinforcement Learning","version":1},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-07T20:08:49.465586Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2502.09923"},"observation_digest":"sha256:1d9a837d53c3d90c4669bfde2792428ee714172fc9f90d2bf8f9c6011d2c50cf","observation_id":"24052ae1-99a5-4399-beeb-237e10d3fe62","resolution":{"observed_at":"2026-08-07T20:08:49.465586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2503.00200","last_updated":"2025-04-24T20:02:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-28T21:38:17Z","title":"Unified Video Action Model","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T17:50:29.675358Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2503.00200"},"observation_digest":"sha256:817be22c5fe0743cfaa4e5fa19b03c89b0243a5ebe333afb179d418316795f80","observation_id":"d561e5f9-dc15-4e00-ab53-803e0186867c","resolution":{"observed_at":"2026-05-13T17:50:29.752426Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-08-07T06:03:33.248255Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06199","last_updated":"2025-06-06T16:00:31Z","snapshot_observed_at":"2026-08-07T16:39:23.915537Z","submitted_at":"2025-06-06T16:00:31Z","title":"3DFlowAction: Learning Cross-Embodiment Manipulation from 3D Flow World Model","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T06:03:33.248255Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2506.06199"},"observation_digest":"sha256:50127a8a3428da40562aea8561df3f0f53d9d66693e9496ca6091b5db88fe30f","observation_id":"038e05b4-7263-4fee-b217-7f22183a200c","resolution":{"observed_at":"2026-08-07T06:03:33.248255Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-08-07T00:25:28.228345Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14198","last_updated":"2025-06-17T05:31:42Z","snapshot_observed_at":"2026-08-08T09:57:20.261631Z","submitted_at":"2025-06-17T05:31:42Z","title":"AMPLIFY: Actionless Motion Priors for Robot Learning from Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T00:25:28.228345Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2506.14198"},"observation_digest":"sha256:694ed519c3f91b518aaa3e197ea259c721e1fc37508dc938533d9a9a3dabb874","observation_id":"a8b07aa3-b76d-49fb-8b1d-8c119778b33f","resolution":{"observed_at":"2026-08-07T00:25:28.228345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-08-06T22:17:13.262031Z","title":"Learning to act from actionless videos through dense correspondences,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22007","last_updated":"2025-06-27T08:21:55Z","snapshot_observed_at":"2026-08-06T22:10:52.001919Z","submitted_at":"2025-06-27T08:21:55Z","title":"RoboEnvision: A Long-Horizon Video Generation Model for Multi-Task Robot Manipulation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:17:13.262031Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2506.22007"},"observation_digest":"sha256:7a68cee23715d9124bff0884cfa2efe6be15c8c5a2b09b9b55eb3e18cacf67e8","observation_id":"b68a66b7-ebec-4d1c-b25d-4d87d2dafbc4","resolution":{"observed_at":"2026-08-06T22:17:13.262031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2507.00990","last_updated":"2026-05-13T01:35:48Z","snapshot_observed_at":"2026-07-06T21:50:35.488353Z","submitted_at":"2025-07-01T17:39:59Z","title":"Robotic Manipulation by Imitating Generated Videos Without Physical Demonstrations","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-19T06:36:13.144868Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2507.00990"},"observation_digest":"sha256:afe0c2dee56a5aabc57e44cacb559cac4af6eed0cbfeef6a6c2b2ea4856172cc","observation_id":"8a768ac1-4892-4d08-9cdf-6ec5cdfac315","resolution":{"observed_at":"2026-05-19T06:37:07.533841Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-08-05T19:15:31.867862Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.13104","last_updated":"2025-08-18T17:12:28Z","snapshot_observed_at":"2026-08-08T21:34:53.802326Z","submitted_at":"2025-08-18T17:12:28Z","title":"Precise Action-to-Video Generation Through Visual Action Prompts","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T19:15:31.867862Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2508.13104"},"observation_digest":"sha256:fa681e871cec3d6eb3713da6737e56fe07f99beb9a685af4c1161a039026d617","observation_id":"f1065572-00dc-42ae-aabf-adf95f136f96","resolution":{"observed_at":"2026-08-05T19:15:31.867862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-08-05T13:46:41.955554Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.00361","last_updated":"2025-08-30T04:53:32Z","snapshot_observed_at":"2026-08-09T09:27:18.513708Z","submitted_at":"2025-08-30T04:53:32Z","title":"Generative Visual Foresight Meets Task-Agnostic Pose Estimation in Robotic Table-Top Manipulation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T13:46:41.955554Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2509.00361"},"observation_digest":"sha256:827aced3b43280481c9591ae210cc195430999e106393301e186fc3aa5dae447","observation_id":"5de35d0b-5b26-46dd-a4a7-6f8f71c7f439","resolution":{"observed_at":"2026-08-05T13:46:41.955554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2512.01773","last_updated":"2026-04-15T17:13:09Z","snapshot_observed_at":"2026-08-03T04:29:49.820139Z","submitted_at":"2025-12-01T15:15:04Z","title":"IGen: Scalable Data Generation for Robot Learning from Open-World Images","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-17T02:58:36.214948Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2512.01773"},"observation_digest":"sha256:9a0a4fc6a3b01f2d6a48e854b55381e8d34ba192cf7d1ea22ff93f799a6dffee","observation_id":"be557eca-5c0d-4c5a-90cd-45f52f3f0f7c","resolution":{"observed_at":"2026-05-17T02:58:54.982338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-08-03T18:09:49.386975Z","title":"Learning to Act from Actionless Videos through Dense Correspondences.arXiv:2310.08576,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.06628","last_updated":"2026-07-05T11:17:31Z","snapshot_observed_at":"2026-08-03T18:09:44.634080Z","submitted_at":"2025-12-07T02:28:06Z","title":"MIND-V: Hierarchical World Model for Long-Horizon Robotic Manipulation with RL-based Physical Alignment","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-03T18:09:49.386975Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2512.06628"},"observation_digest":"sha256:6b3fc10486173d7093703180f17d0536fca1019be2d0bfdef6152b195580ab1a","observation_id":"f627f937-feab-4a91-a22a-6683e1f63285","resolution":{"observed_at":"2026-08-03T18:09:49.386975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2512.15840","last_updated":"2026-05-08T06:37:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-17T18:35:54Z","title":"Large Video Planner Enables Generalizable Robot Control","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-16T21:26:32.048309Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2512.15840"},"observation_digest":"sha256:a64d4328b47083b3478acc248907b14b71ebbfefd2fee714d2d4625d44c2e625","observation_id":"eaa3dc8e-31ce-4fb1-9ffb-ea2304b89dfd","resolution":{"observed_at":"2026-05-16T21:28:33.873937Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-08-03T05:21:36.377047Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.02762","last_updated":"2026-07-02T00:32:14Z","snapshot_observed_at":"2026-08-06T03:52:18.184889Z","submitted_at":"2026-02-02T20:13:43Z","title":"On the Sample Efficiency of Inverse Dynamics Models for Semi-Supervised Imitation Learning","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-03T05:21:36.377047Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2602.02762"},"observation_digest":"sha256:71d9774fd0073a8f9930d769364792f69eb09555559f136f66209664b797f57f","observation_id":"fff2440d-9bb9-4e7c-b318-be2b2652f23c","resolution":{"observed_at":"2026-08-03T05:21:36.377047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2604.03181","last_updated":"2026-04-03T16:57:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-03T16:57:06Z","title":"Multi-View Video Diffusion Policy: A 3D Spatio-Temporal-Aware Video Action Model","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T18:54:07.081457Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2604.03181"},"observation_digest":"sha256:59c54efaf3f44c6bac758b2a268a71ca2e847c26a147214d903402a8ed582562","observation_id":"5a2ec42a-bc32-4d94-940e-41693131aded","resolution":{"observed_at":"2026-05-13T18:58:08.919862Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2604.04974","last_updated":"2026-06-03T17:05:28Z","snapshot_observed_at":"2026-08-05T03:49:14.637609Z","submitted_at":"2026-04-04T15:37:11Z","title":"From Video to Control: A Survey of Learning Manipulation Interfaces from Temporal Visual Data","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-13T17:02:18.358675Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2604.04974"},"observation_digest":"sha256:bf8714e8c2c156a2be75dd8528397e076e40a4767ab5ee2e0e9433ec2fcb22ad","observation_id":"9c0a6822-3700-4022-8b99-9a83c66f8544","resolution":{"observed_at":"2026-05-13T17:03:01.081202Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2604.06168","last_updated":"2026-04-15T01:21:52Z","snapshot_observed_at":"2026-07-06T22:54:44.656835Z","submitted_at":"2026-04-07T17:59:30Z","title":"Action Images: End-to-End Policy Learning via Multiview Video Generation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T18:51:05.206602Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2604.06168"},"observation_digest":"sha256:9174b5d212f67bbff9f2681b0f8b792dedb50f07a5953b778c100cb6517417f4","observation_id":"8a90e2a2-5dcc-41ab-b71c-37bfb2ec9c51","resolution":{"observed_at":"2026-05-10T23:50:53.642876Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2604.11386","last_updated":"2026-04-13T12:25:45Z","snapshot_observed_at":"2026-07-06T22:59:49.614780Z","submitted_at":"2026-04-13T12:25:45Z","title":"ComSim: Building Scalable Real-World Robot Data Generation via Compositional Simulation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:29.558490Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2604.11386"},"observation_digest":"sha256:19856ec191b06866c899583d991e4bceef6ce50c424d220e175cd05275bf41d6","observation_id":"fb396572-8724-4147-8c5f-50bbdb92c5a6","resolution":{"observed_at":"2026-05-11T09:11:01.576249Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2604.15938","last_updated":"2026-04-17T10:56:59Z","snapshot_observed_at":"2026-07-31T03:34:07.969160Z","submitted_at":"2026-04-17T10:56:59Z","title":"VADF: Vision-Adaptive Diffusion Policy Framework for Efficient Robotic Manipulation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T08:31:13.882275Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2604.15938"},"observation_digest":"sha256:543be8299f51ef464602fbfdd9996b5347533911b4145193fb10ecb2f3073b2b","observation_id":"ecd53afa-9db9-4640-934c-4e50eb86db95","resolution":{"observed_at":"2026-05-10T08:32:52.223880Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2604.28185","last_updated":"2026-04-30T17:59:02Z","snapshot_observed_at":"2026-07-06T23:13:29.310140Z","submitted_at":"2026-04-30T17:59:02Z","title":"Visual Generation in the New Era: An Evolution from Atomic Mapping to Agentic World Modeling","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-07T06:38:04.459129Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2604.28185"},"observation_digest":"sha256:7421ae9239b2c3df1329a825ad02d17123343900bd69324d9960ba889a195670","observation_id":"bfa1d90c-26cf-42e0-ade1-c92096b052c7","resolution":{"observed_at":"2026-05-12T10:16:28.801872Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2605.03637","last_updated":"2026-05-05T11:09:41Z","snapshot_observed_at":"2026-08-02T21:37:36.029720Z","submitted_at":"2026-05-05T11:09:41Z","title":"Bridging the Embodiment Gap: Disentangled Cross-Embodiment Video Editing","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-07T15:44:25.021995Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2605.03637"},"observation_digest":"sha256:c2231fd3360d226254332089cf89a134ce50f0cb4c500ba965da5f764e3ba98d","observation_id":"9bf3b0d6-1ab6-4762-80b9-821ca37e7586","resolution":{"observed_at":"2026-05-12T11:01:31.562222Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2605.07474","last_updated":"2026-05-08T09:20:56Z","snapshot_observed_at":"2026-07-06T23:19:51.414344Z","submitted_at":"2026-05-08T09:20:56Z","title":"ForgeVLA: Federated Vision-Language-Action Learning without Language Annotations","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-11T01:51:28.068087Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2605.07474"},"observation_digest":"sha256:b21951406bd7be5f22230c5bc9bd89a09529a0eb65908c9434102e44bdeba570","observation_id":"19837ab5-291c-43bd-9549-e7b647ed0adf","resolution":{"observed_at":"2026-05-11T04:20:55.452734Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2605.12090","last_updated":"2026-05-12T13:10:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-12T13:10:52Z","title":"World Action Models: The Next Frontier in Embodied AI","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-13T05:01:16.802019Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2605.12090"},"observation_digest":"sha256:3288178e451ea6cf498e6a28659bc024fa00906332c28a0fda76cca99c762cf9","observation_id":"610ccdbc-e91b-4615-b336-d6e68066a720","resolution":{"observed_at":"2026-05-13T05:02:17.821723Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2605.12587","last_updated":"2026-05-12T17:59:27Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:27Z","title":"TrackCraft3R: Repurposing Video Diffusion Transformers for Dense 3D Tracking","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-14T21:28:12.151547Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2605.12587"},"observation_digest":"sha256:d1851981db9d66a8353c964ce1acc362f42ef8d0afc707a2aea624739990f6f5","observation_id":"14845897-427a-4a84-836f-04c454d00f3a","resolution":{"observed_at":"2026-05-14T21:29:28.946912Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2606.20781","last_updated":"2026-06-18T17:05:19Z","snapshot_observed_at":"2026-08-03T05:46:19.564000Z","submitted_at":"2026-06-18T17:05:19Z","title":"World Action Models: A Survey","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-06-26T17:11:12.686936Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2606.20781"},"observation_digest":"sha256:a556c0d0a146b1f515ccf4079e347d5f8d4f752fe3333948ee79ca79600da9a0","observation_id":"27f43554-28ff-4d10-9a90-9533a7886926","resolution":{"observed_at":"2026-07-04T04:09:35.263071Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2607.00836","last_updated":"2026-07-25T08:02:03Z","snapshot_observed_at":"2026-08-02T09:17:59.992551Z","submitted_at":"2026-07-01T11:56:54Z","title":"From World Models to World Action Models: A Concise Tutorial for Robotics","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-02T11:26:48.626947Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2607.00836"},"observation_digest":"sha256:c8349a3b2012826727fbb9dc7c01511791db16e37d40a07b4d4b273c8acfbd12","observation_id":"e39ff17c-3763-4afc-996a-7e299882002b","resolution":{"observed_at":"2026-07-02T11:26:53.592737Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2607.00836","last_updated":"2026-07-25T08:02:03Z","snapshot_observed_at":"2026-08-02T09:17:59.992551Z","submitted_at":"2026-07-01T11:56:54Z","title":"From World Models to World Action Models: A Concise Tutorial for Robotics","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-03T20:35:21.512564Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2607.00836"},"observation_digest":"sha256:b7243f3756b4e530f2baf89ebdc95154de035addc5027b1e67a47bb853f0572c","observation_id":"2575fae4-6302-4315-b8a9-346cb712e54e","resolution":{"observed_at":"2026-07-03T20:38:55.067037Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-12T09:15:00.615740Z","title":"Li, L., Zhang, Q., Luo, Y ., Yang, S., Wang, R., Han, F., Yu, M., Gao, Z., Xue, N., Zhu, X., Shen, Y ., and Xu, Y","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.00836","last_updated":"2026-07-25T08:02:03Z","snapshot_observed_at":"2026-08-02T09:17:59.992551Z","submitted_at":"2026-07-01T11:56:54Z","title":"From World Models to World Action Models: A Concise Tutorial for Robotics","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-12T09:15:00.615740Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2607.00836"},"observation_digest":"sha256:c655cf76bb0c832970cd970a70146d71391c97efaf4f491e9d4c68ddbceed163","observation_id":"b9c137da-9f8d-486f-8ef3-1faab3442251","resolution":{"observed_at":"2026-07-12T09:15:00.615740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-08-02T09:18:01.832054Z","title":"Li, L., Zhang, Q., Luo, Y ., Yang, S., Wang, R., Han, F., Yu, M., Gao, Z., Xue, N., Zhu, X., Shen, Y ., and Xu, Y","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.00836","last_updated":"2026-07-25T08:02:03Z","snapshot_observed_at":"2026-08-02T09:17:59.992551Z","submitted_at":"2026-07-01T11:56:54Z","title":"From World Models to World Action Models: A Concise Tutorial for Robotics","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T09:18:01.832054Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2607.00836"},"observation_digest":"sha256:04e98b9f25c99e3d9221cc5e149dde80471be3d9e9a6c145677553a2c7ba3d33","observation_id":"40bfbb25-6c19-4d74-a68b-b17e16312023","resolution":{"observed_at":"2026-08-02T09:18:01.832054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":"2310.08576","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-07-04T04:09:35.261339Z","title":"Learning to act from actionless videos through dense correspondences","venue":null,"work_id":"e2bb8629-96e8-41a3-bfeb-3ece8c4e89f7","year":2024},"citing_paper":{"arxiv_id":"2607.01166","last_updated":"2026-07-01T16:52:49Z","snapshot_observed_at":"2026-08-02T05:07:13.258277Z","submitted_at":"2026-07-01T16:52:49Z","title":"Structured 4D Latent Predictive Model for Robot Planning","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-07-02T11:02:42.282246Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2607.01166"},"observation_digest":"sha256:a681514fa37f387149c20913e4a24a7b103925561d714296c9517e518c416b9c","observation_id":"1c28cbb1-fc72-4ba6-8969-a804d8b5f6f5","resolution":{"observed_at":"2026-07-02T11:06:52.499232Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08576","snapshot_observed_at":"2026-08-01T12:49:03.044032Z","title":"Learning to act from actionless videos through dense correspondences.arXiv preprint arXiv:2310.08576, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19343","last_updated":"2026-07-21T17:59:11Z","snapshot_observed_at":"2026-08-08T07:25:27.980281Z","submitted_at":"2026-07-21T17:59:11Z","title":"Masked Visual Actions for Unified World Modeling","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-01T12:49:03.044032Z"},"links":{"cited_paper":"/paper/2310.08576","citing_paper":"/paper/2607.19343"},"observation_digest":"sha256:a3ed1270ad08bba317dfbe9559d04eb9704988f44a90f127ae8d9b0d56b005bb","observation_id":"d97b8c95-b61e-4c8d-854f-b13625b305e5","resolution":{"observed_at":"2026-08-01T12:49:03.044032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.08576/citation-record","integrity":"/paper/2310.08576/integrity","json":"/paper/2310.08576/citation-record.json","paper":"/paper/2310.08576"},"outbound":[],"paper":{"arxiv_id":"2310.08576","last_updated":"2023-10-12T17:59:23Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-05T14:38:36.869054Z","submitted_at":"2023-10-12T17:59:23Z","title":"Learning to Act from Actionless Videos through Dense Correspondences"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 33 inbound Pith citation observations for arXiv:2310.08576."}