{"as_of":"2026-08-09T06:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a687e14456610bf11f5d1a05e1a8695dc5c62205dbd49a69428e503dffa35800","coverage":[{"denominator":59,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":59,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T00:57:45.185859Z","state":"measured"},{"denominator":59,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":59,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.14852/citation-record","integrity":"/paper/2607.14852/integrity","json":"/paper/2607.14852/citation-record.json","paper":"/paper/2607.14852"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2204.01691","last_updated":"2022-08-16T16:06:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-04T17:57:11Z","title":"Do As I Can, Not As I Say: Grounding Language in Robotic Affordances","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.01691","snapshot_observed_at":"2026-08-02T00:57:38.361430Z","title":"Do as i can, not as i say: Grounding language in robotic affordances.arXiv preprint arXiv:2204.01691, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.361430Z"},"links":{"cited_paper":"/paper/2204.01691","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:323431eba265f86c23a453dfc2a5f0b61c957984cc0d1b07a4974930a8f9240b","observation_id":"6d4e5abc-5875-460f-9143-60c6687f9e05","resolution":{"observed_at":"2026-08-02T00:57:38.361430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.24164","last_updated":"2026-01-08T17:01:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-31T17:22:30Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.24164","snapshot_observed_at":"2026-08-02T00:57:38.446664Z","title":"org/abs/2410.24164, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.446664Z"},"links":{"cited_paper":"/paper/2410.24164","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:d876e8d26585ff273e0d5d002e16d0a5253fde4fd3659b5b3040ebe639761fcf","observation_id":"8da652d4-1b83-4fb1-b00e-29b2ac5e0a2e","resolution":{"observed_at":"2026-08-02T00:57:38.446664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.11706","last_updated":"2023-12-22T13:55:42Z","snapshot_observed_at":"2026-07-06T15:44:42.776459Z","submitted_at":"2023-06-20T17:35:20Z","title":"RoboCat: A Self-Improving Generalist Agent for Robotic Manipulation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.11706","snapshot_observed_at":"2026-08-02T00:57:38.564461Z","title":"Robocat: A self-improving generalist agent for robotic manipulation.arXiv preprint arXiv:2306.11706, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.564461Z"},"links":{"cited_paper":"/paper/2306.11706","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:ec6177938298c7e889686c212047d699c19636fc7f361e8c53c6c545452b13ed","observation_id":"5ecd9849-3188-4f1e-872d-c057bfe1feaf","resolution":{"observed_at":"2026-08-02T00:57:38.564461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06817","last_updated":"2023-08-11T17:45:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-12-13T18:55:15Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.06817","snapshot_observed_at":"2026-08-02T00:57:38.675970Z","title":"Rt-1: Robotics transformer for real-world control at scale.arXiv preprint arXiv:2212.06817, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.675970Z"},"links":{"cited_paper":"/paper/2212.06817","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:1dd88461516870c745f11d95e8e61a55a9a490d6d7d7a0e21618540c41917c42","observation_id":"d23d7b75-13be-4f71-b49f-811a87159e0e","resolution":{"observed_at":"2026-08-02T00:57:38.675970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-02T00:57:38.798291Z","title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control, 2023.URL https://arxiv","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.798291Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:8a2e48e5e26c86447e73a8942d63206e9e5d8da80ff5b521ef30f492f25d4195","observation_id":"bfd3f953-bb97-48a6-8ad8-0565fa985ce2","resolution":{"observed_at":"2026-08-02T00:57:38.798291Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:38.868336Z","title":"Riemannian walk for incremental learning: Understanding forgetting and intransigence","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.868336Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:2e8b894029109cdcc6b945ef642955cbc52838b329e97996a33ea8e0b4b0156c","observation_id":"8fa60eea-fa48-4433-b372-6b24d4ce6af2","resolution":{"observed_at":"2026-08-02T00:57:38.868336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1902.10486","last_updated":"2019-06-04T07:59:35Z","snapshot_observed_at":"2026-08-07T08:44:38.959537Z","submitted_at":"2019-02-27T12:34:19Z","title":"On Tiny Episodic Memories in Continual Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1902.10486","snapshot_observed_at":"2026-08-02T00:57:38.974751Z","title":"On tiny episodic memories in continual learning.arXiv preprint arXiv:1902.10486, 2019","venue":null,"work_id":null,"year":1902},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.974751Z"},"links":{"cited_paper":"/paper/1902.10486","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:4c0fadcfee921c3d98e7635ab802452722e0e652c37eb7b16ededd0d6c7ce679","observation_id":"ac24ec8c-df59-44f7-9361-66b1e17324df","resolution":{"observed_at":"2026-08-02T00:57:38.974751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.04137","last_updated":"2024-03-14T04:36:31Z","snapshot_observed_at":"2026-08-08T17:26:38.829859Z","submitted_at":"2023-03-07T18:50:03Z","title":"Diffusion Policy: Visuomotor Policy Learning via Action Diffusion","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.04137","snapshot_observed_at":"2026-08-02T00:57:39.140456Z","title":"Diffusion policy: Visuomotor policy learning via action diffusion.arXiv preprint arXiv:2303.04137, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.140456Z"},"links":{"cited_paper":"/paper/2303.04137","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:c9fc8e0e4da2be5ce67375d462c396ef95a28f8080060a63016676836e3aaf1a","observation_id":"97243f6c-be38-4b5a-a6fe-ca2c7dec90b3","resolution":{"observed_at":"2026-08-02T00:57:39.140456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03912","last_updated":"2025-05-06T18:35:07Z","snapshot_observed_at":"2026-08-07T15:49:25.272514Z","submitted_at":"2025-05-06T18:35:07Z","title":"OpenHelix: A Short Survey, Empirical Analysis, and Open-Source Dual-System VLA Model for Robotic Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.03912","snapshot_observed_at":"2026-08-02T00:57:39.308136Z","title":"Openhelix: A short survey, empirical analysis, and open-source dual-system vla model for robotic manipulation.arXiv preprint arXiv:2505.03912, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.308136Z"},"links":{"cited_paper":"/paper/2505.03912","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:81397f587549ebff9ce66c2d174856b002bf16b9509b463d97d56e1d0da07def","observation_id":"cf28c9fc-2b4b-4556-8395-f2bd1a2639dc","resolution":{"observed_at":"2026-08-02T00:57:39.308136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:39.473600Z","title":"Loss of plasticity in deep continual learning.Nature, 632(8026):768–774, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.473600Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:d89bcc262645d3b92443bedeea33f79f1b54df816c6e8f75251a8488d8aba6f2","observation_id":"1a53177d-b354-436d-bfe3-879228996277","resolution":{"observed_at":"2026-08-02T00:57:39.473600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:39.612805Z","title":"Palm-e: An embodied multimodal language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.612805Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:ffa927cd945ac5aa7e1683180ab61187baeb6620d71908c34947b32b1bd21ebe","observation_id":"6c03f0c3-b7d1-4ecc-85d8-62fe8f30faf9","resolution":{"observed_at":"2026-08-02T00:57:39.612805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05540","last_updated":"2025-06-17T03:16:46Z","snapshot_observed_at":"2026-08-07T15:48:43.434747Z","submitted_at":"2025-05-08T16:51:36Z","title":"Benchmarking Vision, Language, & Action Models in Procedurally Generated, Open Ended Action Environments","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.05540","snapshot_observed_at":"2026-08-02T00:57:39.775665Z","title":"Benchmarking vision, language, & action models in procedurally generated, open ended action environ- ments.arXiv preprint arXiv:2505.05540, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.775665Z"},"links":{"cited_paper":"/paper/2505.05540","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:fccd7cbac6979d20b980f2ef2a3e2e3a2a9f037a40f2677b01f289164202a577","observation_id":"81402915-0aa4-4d73-9b3f-e82aec51676b","resolution":{"observed_at":"2026-08-02T00:57:39.775665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:39.937349Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.937349Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:8bdf61f72b3cb3e46330daa99471352a7cc968d32d3d29265d262a1b55fb0170","observation_id":"4ba87ad8-c25a-46fe-ae7d-0666ed9f4d4b","resolution":{"observed_at":"2026-08-02T00:57:39.937349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.05973","last_updated":"2023-11-02T06:53:37Z","snapshot_observed_at":"2026-08-05T01:03:23.456778Z","submitted_at":"2023-07-12T07:40:48Z","title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.05973","snapshot_observed_at":"2026-08-02T00:57:40.106650Z","title":"Voxposer: Compos- able 3d value maps for robotic manipulation with language models.arXiv preprint arXiv:2307.05973, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.106650Z"},"links":{"cited_paper":"/paper/2307.05973","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:4434193bf273aebfd78f030ebabd18d0de3c32660c60767e0847b4976e15f736","observation_id":"e5a8d7e6-4f24-4cf7-b19c-973217c16924","resolution":{"observed_at":"2026-08-02T00:57:40.106650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16054","last_updated":"2025-04-22T17:31:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T17:31:29Z","title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16054","snapshot_observed_at":"2026-08-02T00:57:40.272666Z","title":"5: A vision-language-action model with open-world generalization","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.272666Z"},"links":{"cited_paper":"/paper/2504.16054","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:259d4acb2f88499174a5e55c9851d815d186fc90ba25e57b4bb37a02b66b838a","observation_id":"cae95d81-6b39-474d-a382-5de0ee2bbb6e","resolution":{"observed_at":"2026-08-02T00:57:40.272666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03094","last_updated":"2023-05-28T07:32:38Z","snapshot_observed_at":"2026-08-07T13:44:24.778030Z","submitted_at":"2022-10-06T17:50:11Z","title":"VIMA: General Robot Manipulation with Multimodal Prompts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03094","snapshot_observed_at":"2026-08-02T00:57:40.434766Z","title":"Vima: General robot manipulation with multimodal prompts.arXiv preprint arXiv:2210.03094, 2(3):6, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.434766Z"},"links":{"cited_paper":"/paper/2210.03094","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:88a1d846c7c4ddb7c33de54b881f45e11e550fb7172a90b6a534933e8efc60c6","observation_id":"e7c4049a-4dde-4140-8e42-6f4f21c82016","resolution":{"observed_at":"2026-08-02T00:57:40.434766Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:40.557073Z","title":"Srt-h: A hierarchical framework for autonomous surgery via language-conditioned imitation learning.Science robotics, 10(104):eadt5254, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.557073Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:27b7cc4ca945b114f79a001b64251e900e54d5eb947c527fddc14c0f550f38b7","observation_id":"671d6ba0-d7df-499c-9274-8b1d699d88fd","resolution":{"observed_at":"2026-08-02T00:57:40.557073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09246","last_updated":"2024-09-05T19:46:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-13T15:46:55Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09246","snapshot_observed_at":"2026-08-02T00:57:40.564504Z","title":"Openvla: An open-source vision-language- action model.arXiv preprint arXiv:2406.09246, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.564504Z"},"links":{"cited_paper":"/paper/2406.09246","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:ca8302a19034058a06c25d04181115924bf240f0b9c3bf1fc5b74be4a49ef37f","observation_id":"17855fb2-1f2f-45cd-96f9-767e61ea49ca","resolution":{"observed_at":"2026-08-02T00:57:40.564504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:40.626531Z","title":"Overcoming catastrophic forgetting in neural networks.Proceedings of the national academy of sciences, 114(13): 3521–3526, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.626531Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:5355c379bb8e1dcfd9e9c5235ad1fc1fbd7f0298c26f7acf481c258319022a51","observation_id":"7f1747ac-9dad-42dc-b167-da8070caa109","resolution":{"observed_at":"2026-08-02T00:57:40.626531Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.05386","last_updated":"2026-06-29T17:47:49Z","snapshot_observed_at":"2026-08-06T19:25:14.929252Z","submitted_at":"2025-07-07T18:17:06Z","title":"Reinforcement Fine-Tuning Naturally Mitigates Forgetting in Continual Post-Training","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.05386","snapshot_observed_at":"2026-08-02T00:57:40.746075Z","title":"Reinforcement fine-tuning naturally mitigates forgetting in continual post-training.arXiv preprint arXiv:2507.05386, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.746075Z"},"links":{"cited_paper":"/paper/2507.05386","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:dd73d3c15e1a896e76ddad18a6317cd9e62b83ea761965c6e7cbd23fe1ec57a5","observation_id":"973ce0bb-7239-40ee-8772-a2cee4eca53c","resolution":{"observed_at":"2026-08-02T00:57:40.746075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:40.923216Z","title":"The power of scale for parameter-efficient prompt tuning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.923216Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:85cc24b0a7640572e09cead7ed7f0f44763df92c3781b3821d82b1f26cfcc987","observation_id":"920df1b8-d243-4890-b5e7-f1a1efd5aa4b","resolution":{"observed_at":"2026-08-02T00:57:40.923216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.063842Z","title":"Remem-vla: Empowering vision-language-action model with memory via dual-level recurrent queries.arXiv preprint arXiv:2603.12942, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.063842Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:6dbce8f9b014aa4084f960d9ff28c72899afe1dacf3906b230472368c17c350f","observation_id":"1c832505-4417-4d0b-ab31-5a3909d5fd35","resolution":{"observed_at":"2026-08-02T00:57:41.063842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.166042Z","title":"Prefix-tuning: Optimizing continuous prompts for generation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.166042Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:f2f9d8c1ac81e5cb263b177ce7c0cfb2d42edad6bd08da7e2e813330e44024bb","observation_id":"1db9480a-222b-4401-a7ef-7963e83fa241","resolution":{"observed_at":"2026-08-02T00:57:41.166042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.271781Z","title":"Learning without forgetting.IEEE transactions on pattern analysis and machine intelligence, 40(12):2935–2947, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.271781Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:a11e5d56beef7adef0644f5d18465b5fb3e60bdbbc23c773db1fd45dc59dd216","observation_id":"df862715-64fe-47c8-a6f2-2a3e0c02350e","resolution":{"observed_at":"2026-08-02T00:57:41.271781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.376843Z","title":"Learning without forgetting","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.376843Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:44851665d399f34474e42ed3ccbf2f09cc3e4425f8c60af1ed2b3d19b98e2f1e","observation_id":"6d394b78-c309-43d3-aaa3-a0f41b8b07f1","resolution":{"observed_at":"2026-08-02T00:57:41.376843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.519701Z","title":"Code as policies: Language model programs for embodied control","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.519701Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:e8fa45effae2135748b904d8e14dc9e714cd54a4f47cbc200a07a13c7559abba","observation_id":"d083236b-4763-4028-b504-68e83c95dba4","resolution":{"observed_at":"2026-08-02T00:57:41.519701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.679900Z","title":"Never-ending behavior-cloning agent for robotic manipulation.arXiv preprint arXiv:2403.00336, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.679900Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:077f85b80718c2e09d37564709300fd7c93accc0bbe5b41aab520a087eebd8d0","observation_id":"8a6204cf-c9a6-406a-81ed-fce26d6c0c00","resolution":{"observed_at":"2026-08-02T00:57:41.679900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.790039Z","title":"Pixelvla: Advancing pixel-level understanding in vision-language-action model.arXiv preprint arXiv:2511.01571, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.790039Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:b2307567c519f20e0fee64309a60cdfc5b120699d89a6ffce13f5ab180f9af8e","observation_id":"ae038217-681c-40b4-934c-8fbd01cfe4a2","resolution":{"observed_at":"2026-08-02T00:57:41.790039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.20072","last_updated":"2026-05-31T15:50:43Z","snapshot_observed_at":"2026-08-05T15:11:00.360328Z","submitted_at":"2025-08-27T17:39:11Z","title":"Discrete Diffusion VLA: Bringing Discrete Diffusion to Action Decoding in Vision-Language-Action Policies","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.20072","snapshot_observed_at":"2026-08-02T00:57:41.909520Z","title":"Discrete diffusion vla: Bringing discrete diffusion to action decoding in vision-language-action policies.arXiv preprint arXiv:2508.20072, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.909520Z"},"links":{"cited_paper":"/paper/2508.20072","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:4f2210b503c8813cdef18743b157511e1d9183f3c1b1a05cbcdbe47d892b630a","observation_id":"f4e12658-d377-4652-acaf-556b08b76456","resolution":{"observed_at":"2026-08-02T00:57:41.909520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:42.023492Z","title":"Showui: One vision-language-action model for gui visual agent","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.023492Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:dc81231345608528c35f823aefff3fff7eb573f99bc9394cb456f991a019abb6","observation_id":"75d14dfb-e1ee-489d-903a-193620f4e0f4","resolution":{"observed_at":"2026-08-02T00:57:42.023492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:42.163804Z","title":"Libero: Benchmarking knowledge transfer for lifelong robot learning.Advances in Neural Information Processing Systems, 36:44776–44791, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.163804Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:c46cf68b27068175d4fdb709a31c858fb454a954ae1e4bfde2ef4c37a6ee4056","observation_id":"c9dda9e5-a346-4835-9596-a5462827eb2a","resolution":{"observed_at":"2026-08-02T00:57:42.163804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:42.309881Z","title":"Pretrained vision-language-action models are surprisingly resistant to forgetting in continual learning.arXiv preprint arXiv:2603.03818, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.309881Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:f0a9dd4dfcc3e16df315a8d1efb045846cb44430089618043a8f5009b40fba33","observation_id":"2861effb-1148-4105-b9e3-6352c2a770b6","resolution":{"observed_at":"2026-08-02T00:57:42.309881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06710","last_updated":"2025-07-13T06:32:40Z","snapshot_observed_at":"2026-08-06T18:54:09.737742Z","submitted_at":"2025-07-09T10:08:15Z","title":"Spatial-Temporal Aware Visuomotor Diffusion Policy Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06710","snapshot_observed_at":"2026-08-02T00:57:42.384758Z","title":"Spatial- temporal aware visuomotor diffusion policy learning.arXiv preprint arXiv:2507.06710, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.384758Z"},"links":{"cited_paper":"/paper/2507.06710","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:556c1af7d82700d64ae96401711f935b25f64f3712e4747c09e634850f03dd7b","observation_id":"ad0e677d-ee68-4ae8-b596-9ab348fca764","resolution":{"observed_at":"2026-08-02T00:57:42.384758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:42.499558Z","title":"Packnet: Adding multiple tasks to a single network by iterative pruning","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.499558Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:63d1b80078029a77c078e2d3c04584f5c38a083c71d7871e878071d76108a31d","observation_id":"76a1a350-ba0f-4403-b458-12b73c023994","resolution":{"observed_at":"2026-08-02T00:57:42.499558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:42.643648Z","title":"Preserving and combining knowledge in robotic lifelong reinforcement learning.Nature Machine Intelligence, 7(2):256–269, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.643648Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:93ed34566d0c89606d65e61462b243999aa564f5b2574002096935a8e683b0d4","observation_id":"f859b398-a502-45a9-b0de-7c514b6607a0","resolution":{"observed_at":"2026-08-02T00:57:42.643648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11815","last_updated":"2024-06-17T17:55:29Z","snapshot_observed_at":"2026-07-06T18:32:23.999019Z","submitted_at":"2024-06-17T17:55:29Z","title":"LLARVA: Vision-Action Instruction Tuning Enhances Robot Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11815","snapshot_observed_at":"2026-08-02T00:57:42.768376Z","title":"Llarva: Vision-action instruction tuning enhances robot learning.arXiv preprint arXiv:2406.11815, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.768376Z"},"links":{"cited_paper":"/paper/2406.11815","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:723ff5c150b03cca9dad23d090bff63208b1ebea3c5eebfd3d85b20aa9583e39","observation_id":"e21013a9-57c6-4ceb-8be4-35d30bdc961b","resolution":{"observed_at":"2026-08-02T00:57:42.768376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08864","last_updated":"2025-05-14T15:22:36Z","snapshot_observed_at":"2026-08-08T01:53:43.555672Z","submitted_at":"2023-10-13T05:20:40Z","title":"Open X-Embodiment: Robotic Learning Datasets and RT-X Models","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08864","snapshot_observed_at":"2026-08-02T00:57:42.876938Z","title":"Open x-embodiment: Robotic learning datasets and rt-x models.arXiv preprint arXiv:2310.08864, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.876938Z"},"links":{"cited_paper":"/paper/2310.08864","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:36646a4981d8365231f62cd30d2fdb2ad78a547337cc70531775b72b5e8e370e","observation_id":"b7e19fac-c9c1-4f25-801d-4be669680dc4","resolution":{"observed_at":"2026-08-02T00:57:42.876938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09747","last_updated":"2025-01-16T18:57:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-16T18:57:04Z","title":"FAST: Efficient Action Tokenization for Vision-Language-Action Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09747","snapshot_observed_at":"2026-08-02T00:57:42.983225Z","title":"Fast: Efficient action tokenization for vision-language-action models.arXiv preprint arXiv:2501.09747, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.983225Z"},"links":{"cited_paper":"/paper/2501.09747","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:299c8ccc044ab54dd8eb4cafad2044e16cf11bee9339bf0b655e28f213e427bb","observation_id":"8e5de4bf-5370-457c-bf11-394968808e29","resolution":{"observed_at":"2026-08-02T00:57:42.983225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.15483","last_updated":"2026-04-24T23:18:28Z","snapshot_observed_at":"2026-08-05T18:18:04.172934Z","submitted_at":"2026-04-16T19:18:07Z","title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.15483","snapshot_observed_at":"2026-08-02T00:57:43.044566Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.044566Z"},"links":{"cited_paper":"/paper/2604.15483","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:bc0b6b47ecc3caf743bf1d07cf279947e79c969320ca0db167c42ab91f6e2cbf","observation_id":"f093673d-4b63-4894-9d1d-2b3ffc21433c","resolution":{"observed_at":"2026-08-02T00:57:43.044566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:43.119826Z","title":"Incremental classifier and representation learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.119826Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:c9bc4f52a419f8e7bb9b7f6962c96e1cbdd828173d03aec617bf812d72b03480","observation_id":"deffecab-8279-43d4-a60f-555ffe976a3b","resolution":{"observed_at":"2026-08-02T00:57:43.119826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.06175","last_updated":"2022-11-11T10:04:29Z","snapshot_observed_at":"2026-08-08T03:18:33.595658Z","submitted_at":"2022-05-12T16:03:26Z","title":"A Generalist Agent","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.06175","snapshot_observed_at":"2026-08-02T00:57:43.200599Z","title":"A generalist agent.arXiv preprint arXiv:2205.06175, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.200599Z"},"links":{"cited_paper":"/paper/2205.06175","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:80de79b1a9144250d1a5fbcf4139244db0028a535035ce84be395b22cc7fa14a","observation_id":"86645130-7d20-4dd4-838d-caa7f7cfebf8","resolution":{"observed_at":"2026-08-02T00:57:43.200599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1606.04671","last_updated":"2022-10-22T14:34:44Z","snapshot_observed_at":"2026-08-05T07:46:46.355580Z","submitted_at":"2016-06-15T08:20:51Z","title":"Progressive Neural Networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.04671","snapshot_observed_at":"2026-08-02T00:57:43.302391Z","title":"Progressive neural networks.arXiv preprint arXiv:1606.04671, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.302391Z"},"links":{"cited_paper":"/paper/1606.04671","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:941f9c27f41bef644b6c5c927253bac982747489aa382ef1ccfd3db19c3c11fe","observation_id":"e9c286bb-d113-4d4a-bf70-b76e4fcdae19","resolution":{"observed_at":"2026-08-02T00:57:43.302391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.04259","last_updated":"2025-09-04T14:38:08Z","snapshot_observed_at":"2026-08-07T12:32:15.780660Z","submitted_at":"2025-09-04T14:38:08Z","title":"RL's Razor: Why Online Reinforcement Learning Forgets Less","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.04259","snapshot_observed_at":"2026-08-02T00:57:43.368877Z","title":"Rl’s razor: Why online reinforcement learning forgets less.arXiv preprint arXiv:2509.04259, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.368877Z"},"links":{"cited_paper":"/paper/2509.04259","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:49623fa66681cadd6ea8419b57e882e4a7016a0a74b371b1816cbf274b4ae7b0","observation_id":"1d130e34-7e7e-4643-96c8-fc4b7eb9824f","resolution":{"observed_at":"2026-08-02T00:57:43.368877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:43.430502Z","title":"Cliport: What and where pathways for robotic manipulation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.430502Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:8476f92807206c37a041cd3b171278396ae51f7e0fcb1b827c322a175ca728a7","observation_id":"4091b805-2fee-4570-bc78-e314dad59af3","resolution":{"observed_at":"2026-08-02T00:57:43.430502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:43.556355Z","title":"Perceiver-actor: A multi-task transformer for robotic manipulation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.556355Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:dbbfb05c849e87671c18382c6cd372154d5e00641a86e690e45ac8b6c7b7fe6f","observation_id":"30b3ec67-9175-4544-8a41-6fa4ed8db12b","resolution":{"observed_at":"2026-08-02T00:57:43.556355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01844","last_updated":"2025-06-02T16:30:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-02T16:30:19Z","title":"SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.01844","snapshot_observed_at":"2026-08-02T00:57:43.651170Z","title":"Smolvla: A vision-language- action model for affordable and efficient robotics.arXiv preprint arXiv:2506.01844, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.651170Z"},"links":{"cited_paper":"/paper/2506.01844","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:8d3950f71950e01e21bf6a8c4832c8ccc2021e6e2fd0411502a76942865f4d6d","observation_id":"2b2a4d7c-00e1-47e2-a65c-e5d731c23ac4","resolution":{"observed_at":"2026-08-02T00:57:43.651170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:43.723645Z","title":"Coda-prompt: Continual decomposed attention- based prompting for rehearsal-free continual learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.723645Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:862088a99349f330a160bd7869c59c84c179b09fc25584c172e78db2c9d3ff02","observation_id":"68528695-96c2-4fa0-a50c-627e6815c1ab","resolution":{"observed_at":"2026-08-02T00:57:43.723645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.03189","last_updated":"2025-05-30T20:52:21Z","snapshot_observed_at":"2026-08-09T04:57:13.469920Z","submitted_at":"2025-05-30T20:52:21Z","title":"Continual Learning in Vision-Language Models via Aligned Model Merging","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.03189","snapshot_observed_at":"2026-08-02T00:57:43.881234Z","title":"Continual learning in vision-language models via aligned model merging.arXiv preprint arXiv:2506.03189, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.881234Z"},"links":{"cited_paper":"/paper/2506.03189","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:2ecde6fc3cc4738f35a3c93ad47fd3de68a43a22c3361710046d888f8a07d04b","observation_id":"58294f10-753f-46e2-895f-e3ea0d455fe2","resolution":{"observed_at":"2026-08-02T00:57:43.881234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12213","last_updated":"2024-05-26T19:55:26Z","snapshot_observed_at":"2026-07-06T18:16:51.116432Z","submitted_at":"2024-05-20T17:57:01Z","title":"Octo: An Open-Source Generalist Robot Policy","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12213","snapshot_observed_at":"2026-08-02T00:57:44.038690Z","title":"Octo: An open-source generalist robot policy.arXiv preprint arXiv:2405.12213, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.038690Z"},"links":{"cited_paper":"/paper/2405.12213","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:409b313cf549f745b4e574344ad9ff0566e6581e3eb92385535d8e39f5c06af6","observation_id":"3449da03-ae32-4162-bdf3-0cbecd78b540","resolution":{"observed_at":"2026-08-02T00:57:44.038690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:44.188337Z","title":"A comprehensive survey of continual learning: Theory, method and application.IEEE transactions on pattern analysis and machine intelligence, 46(8): 5362–5383, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.188337Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:377ed2e861ab8498bb0a7fa11c4a402641daab3ac7fe038595b7ad14a3a50d50","observation_id":"a5dcc6ca-0af3-4495-8190-012c45d24af2","resolution":{"observed_at":"2026-08-02T00:57:44.188337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.04799","last_updated":"2022-08-05T11:26:06Z","snapshot_observed_at":"2026-08-09T05:58:51.015279Z","submitted_at":"2022-04-10T23:36:55Z","title":"DualPrompt: Complementary Prompting for Rehearsal-free Continual Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.04799","snapshot_observed_at":"2026-08-02T00:57:44.354469Z","title":"Dualprompt: Complementary prompting for rehearsal-free continual learning.arXiv preprint arXiv:2204.04799, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.354469Z"},"links":{"cited_paper":"/paper/2204.04799","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:2fbf64b85b71ff2828485f0c8c39aad34c0fb7c3c5d89c88610de80f1e1d6d2b","observation_id":"697e9aba-0960-4c83-8869-10f31a36f1bd","resolution":{"observed_at":"2026-08-02T00:57:44.354469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:44.474480Z","title":"Tinyvla: Towards fast, data-efficient vision-language-action models for robotic manipulation.IEEE Robotics and Automation Letters, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.474480Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:926269ddf7e896b0171d5b608ae112e724e5fe2e4de85edddeeadcc2a8cd24b9","observation_id":"c29e4a62-2969-42e5-b3e9-618575a75da8","resolution":{"observed_at":"2026-08-02T00:57:44.474480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:44.733577Z","title":"Continual world: A robotic benchmark for continual reinforcement learning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.733577Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:9f23c29bee0cb4f917d1106c7850bc6a233493efb617c006b1569d42a747ca4c","observation_id":"69001641-1349-411b-8022-4794ddb7017b","resolution":{"observed_at":"2026-08-02T00:57:44.733577Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:44.809271Z","title":"Long-horizon language-conditioned imitation learning for robotic manipulation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.809271Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:a59d3802f1b3d07fe4c087770438d5a40eca151b3cf950250f0e9957648f07d5","observation_id":"543c0b8f-1bf6-4ffa-b97d-a34062faa093","resolution":{"observed_at":"2026-08-02T00:57:44.809271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:44.957923Z","title":"Boosting continual learning of vision-language models via mixture-of-experts adapters","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.957923Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:428f8e61db374c49c622126586f0ea83eea0a096c5a9306371571d5b15bf7443","observation_id":"9c3e7bb1-682a-4721-b50b-ea52e0df196e","resolution":{"observed_at":"2026-08-02T00:57:44.957923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:45.006248Z","title":"Atomicvla: Unlocking the potential of atomic skill learning in robots.arXiv preprint arXiv:2603.07648, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:45.006248Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:abf75c36e775b96a8a1ac7ef3775f96c6185dbbc5571475c77d7504d874f3a47","observation_id":"6a83ff64-d458-4670-9690-e4b773621fde","resolution":{"observed_at":"2026-08-02T00:57:45.006248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:45.048809Z","title":"Mllm-cl: Continual learning for multimodal large language models.arXiv preprint arXiv:2506.05453, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:45.048809Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:c21e75c2a122c62e250b72374160ce6d2240a55a4bca72e807ab501101685db0","observation_id":"20ba1cbf-ecb0-48d1-9ff7-2422168cd8fa","resolution":{"observed_at":"2026-08-02T00:57:45.048809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:45.102602Z","title":"Information-theoretic constraints for continual vision-language-action alignment.arXiv preprint arXiv:2603.13335, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:45.102602Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:3b4a89576436c0cf9728adedce4166416fbd51230dd20ea330b0e8e048a8ef77","observation_id":"3bf46d08-2adf-45ed-b526-6ccba4889e8b","resolution":{"observed_at":"2026-08-02T00:57:45.102602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:45.185859Z","title":"imanip: Skill-incremental learning for robotic manipulation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:45.185859Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:5fb3559b6e6f53c118fb5af28fbb07d26067474d9476c076ea4248ff71a3cfea","observation_id":"965b4692-6aed-46b2-801c-c004dcfbc883","resolution":{"observed_at":"2026-08-02T00:57:45.185859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","latest_version":2,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-08T12:29:27.472151Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation"},"reference_resolution":{"displayed":59,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":59,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":59},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 59 of 59 outbound references and 0 inbound Pith citation observations for arXiv:2607.14852."}