{"as_of":"2026-08-16T23:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:23e88153a989de55bf66d3bc792f9a8ad5f37b0d5152972ff14929a6bc7f970e","coverage":[{"denominator":34,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:59:42.014919Z","state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T23:21:12.606042Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09172","snapshot_observed_at":"2026-08-15T23:21:12.606042Z","title":"An open-source software toolkit & benchmark suite for the evaluation and adaptation of multimodal action models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.04921","last_updated":"2025-07-06T14:40:27Z","snapshot_observed_at":"2026-08-15T23:15:46.193892Z","submitted_at":"2025-05-08T03:35:23Z","title":"Perception, Reason, Think, and Plan: A Survey on Large Multimodal Reasoning Models","version":2},"reference_index":153,"source":"arxiv_source","source_observed_at":"2026-08-15T23:21:12.606042Z"},"links":{"cited_paper":"/paper/2506.09172","citing_paper":"/paper/2505.04921"},"observation_digest":"sha256:5131b4200ef4becfc3e7f17137753bbdff030772b8c7033ea1e17680a27c101a","observation_id":"505f9081-c1ea-42bf-b4c0-49afaa9b84fb","resolution":{"observed_at":"2026-08-15T23:21:12.606042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.09172/citation-record","integrity":"/paper/2506.09172/integrity","json":"/paper/2506.09172/citation-record.json","paper":"/paper/2506.09172"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1810.08272","last_updated":"2019-12-19T15:44:33Z","snapshot_observed_at":"2026-08-14T18:12:26.613034Z","submitted_at":"2018-10-18T20:48:08Z","title":"BabyAI: A Platform to Study the Sample Efficiency of Grounded Language Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.08272","snapshot_observed_at":"2026-08-07T04:59:39.925877Z","title":"Clark, P., Cowhey, I., Etzioni, O., Khot, T., Sabharwal, A., Schoenick, C., and Tafjord, O","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:39.925877Z"},"links":{"cited_paper":"/paper/1810.08272","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:55ba7b5e77be3be332609bb60780a2febadcff39c8b157327262d015bb065685","observation_id":"8bd8230d-eec6-49c6-ba37-2385a26eb049","resolution":{"observed_at":"2026-08-07T04:59:39.925877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14108","last_updated":"2023-10-20T17:01:44Z","snapshot_observed_at":"2026-08-16T15:37:17.617814Z","submitted_at":"2023-04-27T11:37:18Z","title":"DataComp: In search of the next generation of multimodal datasets","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14108","snapshot_observed_at":"2026-08-07T04:59:40.265102Z","title":"Gallou´edec, Q., Beeching, E., Romac, C., and Dellandr´ea, E","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.265102Z"},"links":{"cited_paper":"/paper/2304.14108","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:448c6eab36be863c470fce899ae17508246d446fd261be90625a6a5a6568610e","observation_id":"28da12cc-c86a-4b77-a6b4-f5ab6e92c7cb","resolution":{"observed_at":"2026-08-07T04:59:40.265102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.09844","last_updated":"2024-07-10T15:56:14Z","snapshot_observed_at":"2026-08-16T14:18:22.084265Z","submitted_at":"2024-02-15T10:01:55Z","title":"Jack of All Trades, Master of Some, a Multi-Purpose Transformer Agent","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.09844","snapshot_observed_at":"2026-08-07T04:59:40.350755Z","title":"Goyal, Y ., Khot, T., Summers-Stay, D., Batra, D., and Parikh, D","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.350755Z"},"links":{"cited_paper":"/paper/2402.09844","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:b0ce2cb1218cf3ac93599d5b2ffaa1a0c2b454530856e04ef5a0ee69df8407ec","observation_id":"961607ac-4284-4ff4-90f0-f81335d3d48f","resolution":{"observed_at":"2026-08-07T04:59:40.350755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.13888","last_updated":"2021-02-12T18:34:07Z","snapshot_observed_at":"2026-08-15T10:25:01.119307Z","submitted_at":"2020-06-24T17:14:51Z","title":"RL Unplugged: A Suite of Benchmarks for Offline Reinforcement Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.13888","snapshot_observed_at":"2026-08-07T04:59:40.413423Z","title":"Gurari, D., Li, Q., Stangl, A","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.413423Z"},"links":{"cited_paper":"/paper/2006.13888","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:13d89485b06a36987250eae8b813624dcc51df18ded2d598c7f564cf8f2678c0","observation_id":"00413ff4-682f-460d-ab2b-6998e47efa74","resolution":{"observed_at":"2026-08-07T04:59:40.413423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.08218","last_updated":"2018-05-09T17:26:40Z","snapshot_observed_at":"2026-08-14T19:43:11.278539Z","submitted_at":"2018-02-22T18:16:53Z","title":"VizWiz Grand Challenge: Answering Visual Questions from Blind People","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.08218","snapshot_observed_at":"2026-08-07T04:59:40.482183Z","title":"Guruprasad, P., Sikka, H., Song, J., Wang, Y ., and Liang, P","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.482183Z"},"links":{"cited_paper":"/paper/1802.08218","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:157f37774ecafbcf9fbaee476501b94645a86a5982bec4324c2325c244b28f06","observation_id":"59804616-2247-4fc0-aa48-e61f3ee5a5b2","resolution":{"observed_at":"2026-08-07T04:59:40.482183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.05821","last_updated":"2024-12-08T06:54:43Z","snapshot_observed_at":"2026-08-16T13:03:04.430366Z","submitted_at":"2024-11-04T18:01:34Z","title":"Benchmarking Vision, Language, & Action Models on Robotic Learning Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.05821","snapshot_observed_at":"2026-08-07T04:59:40.554768Z","title":"org/abs/2411.05821","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.554768Z"},"links":{"cited_paper":"/paper/2411.05821","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:24bc5d24743b0414dc3d1b4010a9c24d834e3032a23a3168e59143be6aa067a0","observation_id":"57d0d3b1-cf47-4220-ab74-496fcaffe9c4","resolution":{"observed_at":"2026-08-07T04:59:40.554768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09747","last_updated":"2025-01-16T18:57:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-16T18:57:04Z","title":"FAST: Efficient Action Tokenization for Vision-Language-Action Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09747","snapshot_observed_at":"2026-08-07T04:59:40.697308Z","title":"org/abs/2501.09747","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.697308Z"},"links":{"cited_paper":"/paper/2501.09747","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:6c833a9369be6edd0b8f51fba43ad4ccd4bff8afff62e228bfedf825247b4338","observation_id":"c68dfd8f-8ce0-4313-b9df-371acec99d58","resolution":{"observed_at":"2026-08-07T04:59:40.697308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.16527","last_updated":"2023-08-21T09:35:52Z","snapshot_observed_at":"2026-08-16T15:22:13.511618Z","submitted_at":"2023-06-21T14:01:01Z","title":"OBELICS: An Open Web-Scale Filtered Dataset of Interleaved Image-Text Documents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.16527","snapshot_observed_at":"2026-08-07T04:59:40.801771Z","title":"Liang, P","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.801771Z"},"links":{"cited_paper":"/paper/2306.16527","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:7c10139da9196be4be377a09e030344701fec441ad5c399c9696b9c02190f913","observation_id":"5fafa193-8c18-4706-89f9-cfaaa33f3fc3","resolution":{"observed_at":"2026-08-07T04:59:40.801771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.07502","last_updated":"2021-11-10T07:31:56Z","snapshot_observed_at":"2026-08-16T20:17:20.317174Z","submitted_at":"2021-07-15T17:54:36Z","title":"MultiBench: Multiscale Benchmarks for Multimodal Representation Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.07502","snapshot_observed_at":"2026-08-07T04:59:40.875957Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.875957Z"},"links":{"cited_paper":"/paper/2107.07502","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:dbc163b4e3c3bff87bde4587a1d1ac416f3f517307f9d9a7623a6b48471885b6","observation_id":"5a409a04-29cb-4b8c-9a98-b3240d0de6cc","resolution":{"observed_at":"2026-08-07T04:59:40.875957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04779","last_updated":"2023-07-06T16:46:11Z","snapshot_observed_at":"2026-08-16T16:54:03.141085Z","submitted_at":"2022-06-09T22:08:47Z","title":"Challenges and Opportunities in Offline Reinforcement Learning from Visual Observations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.04779","snapshot_observed_at":"2026-08-07T04:59:40.954327Z","title":"Penedo, G., Kydl´ıˇcek, H., allal, L","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.954327Z"},"links":{"cited_paper":"/paper/2206.04779","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:cfa2e9131454b1f8fd247c377234f6cf26fc8dbc34e2bca71803a9e1bfcb0bdc","observation_id":"1d1ef4f6-74b2-439d-920f-f0373164c067","resolution":{"observed_at":"2026-08-07T04:59:40.954327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17557","last_updated":"2024-10-31T11:37:49Z","snapshot_observed_at":"2026-08-14T14:43:02.130172Z","submitted_at":"2024-06-25T13:50:56Z","title":"The FineWeb Datasets: Decanting the Web for the Finest Text Data at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17557","snapshot_observed_at":"2026-08-07T04:59:41.031497Z","title":"Plummer, B","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.031497Z"},"links":{"cited_paper":"/paper/2406.17557","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:5d3817bfecc20bb41332da100912644ec231bf40b3521ad671bc0bcb5a2160ad","observation_id":"45e50e57-3e25-44aa-9044-ff76479cb3fa","resolution":{"observed_at":"2026-08-07T04:59:41.031497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1505.04870","last_updated":"2016-09-19T20:20:42Z","snapshot_observed_at":"2026-08-14T22:48:03.209909Z","submitted_at":"2015-05-19T04:46:03Z","title":"Flickr30k Entities: Collecting Region-to-Phrase Correspondences for Richer Image-to-Sentence Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1505.04870","snapshot_observed_at":"2026-08-07T04:59:41.110128Z","title":"Schwenk, D., Khandelwal, A., Clark, C., Marino, K., and Mottaghi, R","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.110128Z"},"links":{"cited_paper":"/paper/1505.04870","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:411ce15468778d3b0478f30b1e532257779ad60258d474cace264a19068b2a12","observation_id":"2f61d586-ef08-44ce-8b03-c26ca978f2b0","resolution":{"observed_at":"2026-08-07T04:59:41.110128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.01718","last_updated":"2022-06-03T17:52:27Z","snapshot_observed_at":"2026-08-16T16:55:26.992897Z","submitted_at":"2022-06-03T17:52:27Z","title":"A-OKVQA: A Benchmark for Visual Question Answering using World Knowledge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.01718","snapshot_observed_at":"2026-08-07T04:59:41.169719Z","title":"Sharma, P., Ding, N., Goodman, S., and Soricut, R","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.169719Z"},"links":{"cited_paper":"/paper/2206.01718","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:98dfcd795efc48a1e407e0a6833c845654fe7375e01ccf68ec138c6f1ca8fa64","observation_id":"c3dd53f6-5450-4264-ba80-d621e0ba4f29","resolution":{"observed_at":"2026-08-07T04:59:41.169719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:59:41.236551Z","title":"doi: 10.18653/v1/P18-1238","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.236551Z"},"links":{"citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:4ac27f2a35cc08b40643ca259ebd55661d48cfda7c9826dc22810c74a61bb96b","observation_id":"be61615c-d28c-4cf3-b9f6-0c59d0a9debe","resolution":{"observed_at":"2026-08-07T04:59:41.236551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1811.00937","last_updated":"2019-03-15T18:02:58Z","snapshot_observed_at":"2026-08-14T18:05:06.623407Z","submitted_at":"2018-11-02T15:34:29Z","title":"CommonsenseQA: A Question Answering Challenge Targeting Commonsense Knowledge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1811.00937","snapshot_observed_at":"2026-08-07T04:59:41.295786Z","title":"Tassa, Y ., Doron, Y ., Muldal, A., Erez, T., Li, Y ., de Las Casas, D., Budden, D., Abdolmaleki, A., Merel, J., Lefrancq, A., Lillicrap, T., and Riedmiller, M","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.295786Z"},"links":{"cited_paper":"/paper/1811.00937","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:7bcae06f672a7a3803c46205ddb613b77e4a19de33eef112b190e7e756d7dac5","observation_id":"97b11d65-ca8b-4f5b-ba92-ab97f8877861","resolution":{"observed_at":"2026-08-07T04:59:41.295786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1801.00690","last_updated":"2018-01-02T15:48:14Z","snapshot_observed_at":"2026-08-01T20:24:08.300098Z","submitted_at":"2018-01-02T15:48:14Z","title":"DeepMind Control Suite","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1801.00690","snapshot_observed_at":"2026-08-07T04:59:41.401667Z","title":"Todorov, E., Erez, T., and Tassa, Y","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.401667Z"},"links":{"cited_paper":"/paper/1801.00690","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:e60c991a0f2356d85c916f63b8638170c974b7324c478fcd55843d7e768b0f11","observation_id":"60bf0a01-09cd-4194-869f-ddc7a0b9ec41","resolution":{"observed_at":"2026-08-07T04:59:41.401667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T04:59:41.574544Z","title":"Vedantam, R., Zitnick, C","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.574544Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:7b794b8fb100c8d09ea87dae8d1e6c7fd594bcaa3c3cb5488e3812c703cba9d7","observation_id":"cdb030db-b583-4598-97be-3374417e8e7d","resolution":{"observed_at":"2026-08-07T04:59:41.574544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19314","snapshot_observed_at":"2026-08-07T04:59:41.787905Z","title":"org/abs/2406.19314","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.787905Z"},"links":{"cited_paper":"/paper/2406.19314","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:f97e4b6c92dac861ed2b3f01e7181ed276c4f3922ec6e3c2234c79f2459c490c","observation_id":"bc3d99d6-9615-4590-88d6-7e3b9bcc20eb","resolution":{"observed_at":"2026-08-07T04:59:41.787905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.10897","last_updated":"2021-06-14T18:45:16Z","snapshot_observed_at":"2026-08-14T00:30:45.171423Z","submitted_at":"2019-10-24T03:19:46Z","title":"Meta-World: A Benchmark and Evaluation for Multi-Task and Meta Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.10897","snapshot_observed_at":"2026-08-07T04:59:41.871352Z","title":"Zellers, R., Holtzman, A., Bisk, Y ., Farhadi, A., and Choi, Y","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.871352Z"},"links":{"cited_paper":"/paper/1910.10897","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:522a2e0d1a074882ccd4eae8ecb26c6646a7da04b7d214b902b4577004673fe1","observation_id":"a2e551cf-2fb8-40fc-a387-d64f945924d8","resolution":{"observed_at":"2026-08-07T04:59:41.871352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.07830","last_updated":"2019-05-19T23:57:23Z","snapshot_observed_at":"2026-08-15T09:37:44.321271Z","submitted_at":"2019-05-19T23:57:23Z","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.07830","snapshot_observed_at":"2026-08-07T04:59:41.955558Z","title":"Zhai, X., Mustafa, B., Kolesnikov, A., and Beyer, L","venue":null,"work_id":null,"year":1905},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.955558Z"},"links":{"cited_paper":"/paper/1905.07830","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:17aca68ab18bb0f5c060c1a9efcf736585c8201c2b402f58ff35b4b7ee6e13eb","observation_id":"d4df784a-6744-4d5f-a114-cf5a44dd2e69","resolution":{"observed_at":"2026-08-07T04:59:41.955558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15343","last_updated":"2023-09-27T12:05:41Z","snapshot_observed_at":"2026-07-06T15:08:30.190912Z","submitted_at":"2023-03-27T15:53:01Z","title":"Sigmoid Loss for Language Image Pre-Training","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.15343","snapshot_observed_at":"2026-08-07T04:59:42.014919Z","title":"7 MultiNet: An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models A","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:42.014919Z"},"links":{"cited_paper":"/paper/2303.15343","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:d4ddb0ff3398dc68cefc09096dd47798642dcc76e3de38bc19adcf22ddcb6b8c","observation_id":"393baf8c-d3a7-4b6c-9d3b-830fe3672737","resolution":{"observed_at":"2026-08-07T04:59:42.014919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:59:39.852900Z","title":null,"venue":null,"work_id":null,"year":1950},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":1950,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:39.852900Z"},"links":{"citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:bd1d3519b3b04ec97c92af21225296658b202a09e4ca673a85ded38c47dbe1ef","observation_id":"cf28058c-b84f-4c16-9dd3-3188c89cf1f6","resolution":{"observed_at":"2026-08-07T04:59:39.852900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:59:41.468806Z","title":"Touvron, H., Martin, L., Stone, K., Albert, P., Almahairi, A., Babaei, Y ., Bashlykov, N., Batra, S., Bhargava, P., Bhosale, S., Bikel, D., Blecher, L., Ferrer, C","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2012,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.468806Z"},"links":{"citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:17ef9577a64a524c11b6661a27c5dcd4d55ae434007add7f56d4be5b20641765","observation_id":"70fb49f0-8766-417a-87a9-6c47cddfc19d","resolution":{"observed_at":"2026-08-07T04:59:41.468806Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:59:39.606501Z","title":"doi: 10.1613/jair.3912","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2013,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:39.606501Z"},"links":{"citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:829cb6c4281629b87c2c33bea8ebf82df9eb6f792d8cd42c07c8c2d480c841f4","observation_id":"d69aa80a-1219-409f-aa1b-629d11665949","resolution":{"observed_at":"2026-08-07T04:59:39.606501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1411.5726","last_updated":"2015-06-03T01:42:20Z","snapshot_observed_at":"2026-08-14T23:10:48.763001Z","submitted_at":"2014-11-20T23:54:35Z","title":"CIDEr: Consensus-based Image Description Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1411.5726","snapshot_observed_at":"2026-08-07T04:59:41.668726Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:41.668726Z"},"links":{"cited_paper":"/paper/1411.5726","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:7ccb0e6ed2855ddfa91c30929f3543f919f6ef67dc81bbb44dab202efbf1629f","observation_id":"f9f865c2-abb9-4939-9939-08c2c8d6d90c","resolution":{"observed_at":"2026-08-07T04:59:41.668726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.03801","last_updated":"2016-12-13T12:19:48Z","snapshot_observed_at":"2026-08-14T21:25:55.714417Z","submitted_at":"2016-12-12T17:32:49Z","title":"DeepMind Lab","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.03801","snapshot_observed_at":"2026-08-07T04:59:39.552952Z","title":"Bellemare, M","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:39.552952Z"},"links":{"cited_paper":"/paper/1612.03801","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:295d81847845af5ccf04b4e4e46908af460a90cc3df0011d7dda593b66b0d972","observation_id":"3028a8e2-c23b-4261-9830-88d5a8ddceb5","resolution":{"observed_at":"2026-08-07T04:59:39.552952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-08-14T19:36:07.505691Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-08-07T04:59:40.023355Z","title":"Cobbe, K., Hesse, C., Hilton, J., and Schulman, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.023355Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:69faca25e216dfc9c36d0e501b0ba862c08ae61a503fd87416c2dc4214b0ecb1","observation_id":"f0e71d61-a184-4917-b8e0-331e40199d63","resolution":{"observed_at":"2026-08-07T04:59:40.023355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:59:42.551973Z","title":"cc/paper_files/paper/2019/file/ 97af07a14cacba681feacf3012730892-Paper","venue":null,"work_id":"62e4414d-a83e-4c6a-b536-688df0a81eb6","year":2019},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:39.506465Z"},"links":{"citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:26b4e9644d6a8341a93d0a89afe075aeba12ca9d3f1c15addf6e99e4f893b061","observation_id":"7fc9a80d-f984-4a32-abd9-da391424417a","resolution":{"observed_at":"2026-08-07T04:59:42.604335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1912.01588","last_updated":"2020-07-26T18:39:26Z","snapshot_observed_at":"2026-08-06T18:05:01.923590Z","submitted_at":"2019-12-03T18:34:03Z","title":"Leveraging Procedural Generation to Benchmark Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.01588","snapshot_observed_at":"2026-08-07T04:59:40.090855Z","title":"Collaboration, O","venue":null,"work_id":null,"year":1912},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.090855Z"},"links":{"cited_paper":"/paper/1912.01588","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:3e19a0ab758ff59ed252e38ba6f3ea7561e7dbffcacb5bb6256fe5aa7a0f0686","observation_id":"b42d011d-d382-4b29-ae7a-c14f333f4633","resolution":{"observed_at":"2026-08-07T04:59:40.090855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2004.07219","last_updated":"2021-02-06T01:57:28Z","snapshot_observed_at":"2026-08-16T08:32:46.407746Z","submitted_at":"2020-04-15T17:18:19Z","title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.07219","snapshot_observed_at":"2026-08-07T04:59:40.186796Z","title":"Gadre, S","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.186796Z"},"links":{"cited_paper":"/paper/2004.07219","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:c953b79364445d57f67c3947c9203da2df7ca08d3722cd202874bd12df3bb6e1","observation_id":"3c807595-8d77-4bd4-8053-f293b4075edf","resolution":{"observed_at":"2026-08-07T04:59:40.186796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.12576","last_updated":"2022-10-11T13:59:53Z","snapshot_observed_at":"2026-08-16T16:43:24.872712Z","submitted_at":"2022-07-25T23:57:44Z","title":"WinoGAViL: Gamified Association Benchmark to Challenge Vision-and-Language Models","version":2},"cited_work":{"arxiv_id":"2207.12576","doi":null,"metadata_source":"pith","pith_arxiv_id":"2207.12576","snapshot_observed_at":"2026-08-07T04:59:42.397399Z","title":"WinoGAViL: Gamified Association Benchmark to Challenge Vision-and-Language Models","venue":"cs.CL","work_id":"e5f1e87b-edad-4179-bc28-58ec3c02d625","year":2022},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:39.702725Z"},"links":{"cited_paper":"/paper/2207.12576","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:f640ee1eb5e545b331772d556f4b093bcada99473f540b10a287165fa1d0d740","observation_id":"6e89ba29-2023-4d2c-87c6-664b7eb14d10","resolution":{"observed_at":"2026-08-07T04:59:42.444373Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.02496","last_updated":"2023-11-30T17:47:04Z","snapshot_observed_at":"2026-08-16T14:45:58.483915Z","submitted_at":"2023-11-04T19:41:50Z","title":"LocoMuJoCo: A Comprehensive Imitation Learning Benchmark for Locomotion","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.02496","snapshot_observed_at":"2026-08-07T04:59:39.454884Z","title":"Barbu, A., Mayo, D., Alverio, J., Luo, W., Wang, C., Gutfreund, D., Tenenbaum, J., and Katz, B","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:39.454884Z"},"links":{"cited_paper":"/paper/2311.02496","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:cf2cd78bdd8266e5b665083e53ebc6d8d13177fdbab937a23695f074feb46d95","observation_id":"d5a9cc09-a582-4548-8d11-e0903d92174b","resolution":{"observed_at":"2026-08-07T04:59:39.454884Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.24164","last_updated":"2026-01-08T17:01:05Z","snapshot_observed_at":"2026-08-16T17:53:54.636855Z","submitted_at":"2024-10-31T17:22:30Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.24164","snapshot_observed_at":"2026-08-07T04:59:39.772360Z","title":"org/abs/2410.24164","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:39.772360Z"},"links":{"cited_paper":"/paper/2410.24164","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:d11bb3dbab07181d5f2e58bd41a8e3fbe13dc3580151b11d5330291230e75cc3","observation_id":"c0b6eb64-ace5-4ecc-96be-0ef63c2fd397","resolution":{"observed_at":"2026-08-07T04:59:39.772360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05540","last_updated":"2025-06-17T03:16:46Z","snapshot_observed_at":"2026-08-16T00:06:05.331251Z","submitted_at":"2025-05-08T16:51:36Z","title":"Benchmarking Vision, Language, & Action Models in Procedurally Generated, Open Ended Action Environments","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.05540","snapshot_observed_at":"2026-08-07T04:59:40.636188Z","title":"Hendrycks, D., Basart, S., Mu, N., Kadavath, S., Wang, F., Dorundo, E., Desai, R., Zhu, T., Parajuli, S., Guo, M., Song, D., Steinhardt, J., and Gilmer, J","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T04:59:40.636188Z"},"links":{"cited_paper":"/paper/2505.05540","citing_paper":"/paper/2506.09172"},"observation_digest":"sha256:bd5e932b99f8bc1f084ed621a871bebf3fdcdd830d31cd41137fd8360f87b010","observation_id":"dd544236-06b5-4bf9-be0c-79b7b3fc0536","resolution":{"observed_at":"2026-08-07T04:59:40.636188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.09172","last_updated":"2025-06-17T03:30:48Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-10T04:09:06.138790Z","submitted_at":"2025-06-10T18:38:19Z","title":"An Open-Source Software Toolkit & Benchmark Suite for the Evaluation and Adaptation of Multimodal Action Models"},"reference_resolution":{"displayed":34,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":32,"verified_exact":1,"verified_fuzzy":1},"total_outbound_references":34},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 34 of 34 outbound references and 1 inbound Pith citation observation for arXiv:2506.09172."}