{"as_of":"2026-08-09T19:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:686bdd90295137630f4bd9a615069cc8939df14767b63d8825f4d00a36ea5b6a","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":6,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":6,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":6,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:42:16.234764Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T19:30:07.232112Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.02071","last_updated":"2023-11-28T11:23:14Z","snapshot_observed_at":"2026-08-09T10:02:13.123149Z","submitted_at":"2023-10-03T14:13:36Z","title":"Towards End-to-End Embodied Decision Making via Multi-modal Large Language Model: Explorations with GPT4-Vision and Beyond","version":4},"cited_work":{"arxiv_id":"2310.02071","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02071","snapshot_observed_at":"2026-07-04T19:30:07.232112Z","title":"Towards end-to-end embodied decision making via multi-modal large language model: Explorations with gpt4-vision and beyond","venue":null,"work_id":"4e91aa8f-0530-48b5-8a4a-6b1ff965b358","year":2023},"citing_paper":{"arxiv_id":"2312.08935","last_updated":"2024-02-19T14:07:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-14T13:41:54Z","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","version":3},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-14T22:34:15.638114Z"},"links":{"cited_paper":"/paper/2310.02071","citing_paper":"/paper/2312.08935"},"observation_digest":"sha256:20a2d1000ab141e39461e43be146dbca48b1bbc661cd23ae729b8526c63625f0","observation_id":"7de0c3b4-2a71-4710-9273-fd710dfd9103","resolution":{"observed_at":"2026-05-14T22:34:15.938776Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02071","last_updated":"2023-11-28T11:23:14Z","snapshot_observed_at":"2026-08-09T10:02:13.123149Z","submitted_at":"2023-10-03T14:13:36Z","title":"Towards End-to-End Embodied Decision Making via Multi-modal Large Language Model: Explorations with GPT4-Vision and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02071","snapshot_observed_at":"2026-08-07T12:42:16.234764Z","title":"Towards end-to-end embodied decision making via multi- modal large language model: Explorations with gpt4-vision and beyond","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23977","last_updated":"2025-05-29T20:08:36Z","snapshot_observed_at":"2026-08-09T14:39:26.249780Z","submitted_at":"2025-05-29T20:08:36Z","title":"VisualSphinx: Large-Scale Synthetic Vision Logic Puzzles for RL","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T12:42:16.234764Z"},"links":{"cited_paper":"/paper/2310.02071","citing_paper":"/paper/2505.23977"},"observation_digest":"sha256:a9f7f729289645fb6c9842e9e792cc5b0f9a1ccffffba02d5507ea6ce896d68a","observation_id":"17ea7b7e-b8fd-4e11-8470-9104513acf82","resolution":{"observed_at":"2026-08-07T12:42:16.234764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02071","last_updated":"2023-11-28T11:23:14Z","snapshot_observed_at":"2026-08-09T10:02:13.123149Z","submitted_at":"2023-10-03T14:13:36Z","title":"Towards End-to-End Embodied Decision Making via Multi-modal Large Language Model: Explorations with GPT4-Vision and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02071","snapshot_observed_at":"2026-08-06T22:36:42.313802Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21252","last_updated":"2025-06-26T13:36:12Z","snapshot_observed_at":"2026-08-07T12:39:14.206067Z","submitted_at":"2025-06-26T13:36:12Z","title":"Agent-RewardBench: Towards a Unified Benchmark for Reward Modeling across Perception, Planning, and Safety in Real-World Multimodal Agents","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T22:36:42.313802Z"},"links":{"cited_paper":"/paper/2310.02071","citing_paper":"/paper/2506.21252"},"observation_digest":"sha256:a6752398fdbfee84d1e775b77fb69bd3e52b9618c9a7e0c99d03b2f3e54b1dde","observation_id":"2d722460-63ac-4ef0-bdcd-672c1891413e","resolution":{"observed_at":"2026-08-06T22:36:42.313802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02071","last_updated":"2023-11-28T11:23:14Z","snapshot_observed_at":"2026-08-09T10:02:13.123149Z","submitted_at":"2023-10-03T14:13:36Z","title":"Towards End-to-End Embodied Decision Making via Multi-modal Large Language Model: Explorations with GPT4-Vision and Beyond","version":4},"cited_work":{"arxiv_id":"2310.02071","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02071","snapshot_observed_at":"2026-07-04T19:30:07.232112Z","title":"Towards end-to-end embodied decision making via multi-modal large language model: Explorations with gpt4-vision and beyond","venue":null,"work_id":"4e91aa8f-0530-48b5-8a4a-6b1ff965b358","year":2023},"citing_paper":{"arxiv_id":"2509.07553","last_updated":"2026-04-03T02:36:28Z","snapshot_observed_at":"2026-07-06T22:26:22.911328Z","submitted_at":"2025-09-09T09:46:01Z","title":"VeriOS: Query-Driven Proactive Human-Agent-GUI Interaction for Trustworthy OS Agents","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-18T18:06:12.349285Z"},"links":{"cited_paper":"/paper/2310.02071","citing_paper":"/paper/2509.07553"},"observation_digest":"sha256:e46fd7461b1b5d4bd84413fabcddb2ffe90a95402de60e6a7b7146add1704a42","observation_id":"391f5fc7-8137-4a86-810f-9b9ef62626dc","resolution":{"observed_at":"2026-05-18T18:06:42.715464Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02071","last_updated":"2023-11-28T11:23:14Z","snapshot_observed_at":"2026-08-09T10:02:13.123149Z","submitted_at":"2023-10-03T14:13:36Z","title":"Towards End-to-End Embodied Decision Making via Multi-modal Large Language Model: Explorations with GPT4-Vision and Beyond","version":4},"cited_work":{"arxiv_id":"2310.02071","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02071","snapshot_observed_at":"2026-07-04T19:30:07.232112Z","title":"Towards end-to-end embodied decision making via multi-modal large language model: Explorations with gpt4-vision and beyond","venue":null,"work_id":"4e91aa8f-0530-48b5-8a4a-6b1ff965b358","year":2023},"citing_paper":{"arxiv_id":"2606.25509","last_updated":"2026-06-24T07:44:51Z","snapshot_observed_at":"2026-07-06T23:59:57.346004Z","submitted_at":"2026-06-24T07:44:51Z","title":"ASSCG: Just-Right Gating over Chattering for Fast-Slow LLM Planning in Autonomous Driving","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-25T21:20:14.654132Z"},"links":{"cited_paper":"/paper/2310.02071","citing_paper":"/paper/2606.25509"},"observation_digest":"sha256:d8ec0569fdf1df7d203de44ff94f04e04904f8f5f98684f1be6fa234277bf60b","observation_id":"891377d2-ed42-4360-a8f0-76f793a5c286","resolution":{"observed_at":"2026-07-04T19:30:07.233474Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02071","last_updated":"2023-11-28T11:23:14Z","snapshot_observed_at":"2026-08-09T10:02:13.123149Z","submitted_at":"2023-10-03T14:13:36Z","title":"Towards End-to-End Embodied Decision Making via Multi-modal Large Language Model: Explorations with GPT4-Vision and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02071","snapshot_observed_at":"2026-07-31T21:55:17.407982Z","title":"arXiv preprint arXiv:2310.02071 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27967","last_updated":"2026-07-30T10:14:24Z","snapshot_observed_at":"2026-08-05T12:14:09.489312Z","submitted_at":"2026-07-30T10:14:24Z","title":"MARS-RA: Rank Aggregation for Credit Assignment via Multimodal Comparisons in Embodied Multi-Agent Cooperation","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-07-31T21:55:17.407982Z"},"links":{"cited_paper":"/paper/2310.02071","citing_paper":"/paper/2607.27967"},"observation_digest":"sha256:e6d2e02d37856697d488e312c8333fa6cfee108e9952596527b71106b5ce47ad","observation_id":"f2a2672f-19d1-4f01-943d-2249ddd162fe","resolution":{"observed_at":"2026-07-31T21:55:17.407982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.02071/citation-record","integrity":"/paper/2310.02071/integrity","json":"/paper/2310.02071/citation-record.json","paper":"/paper/2310.02071"},"outbound":[],"paper":{"arxiv_id":"2310.02071","last_updated":"2023-11-28T11:23:14Z","latest_version":4,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T10:02:13.123149Z","submitted_at":"2023-10-03T14:13:36Z","title":"Towards End-to-End Embodied Decision Making via Multi-modal Large Language Model: Explorations with GPT4-Vision and Beyond"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 6 inbound Pith citation observations for arXiv:2310.02071."}