{"as_of":"2026-08-07T07:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:90c6b13350522891d8c7dc76dab3ec0cbd231e5816b0ae072d343eef4cc8bfe3","coverage":[{"denominator":41,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T04:08:14.068241Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T01:10:21.243699Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-04T13:29:51.596096Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.05843","snapshot_observed_at":"2026-07-13T10:21:59.840828Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2604.04266","last_updated":"2026-04-05T21:08:00Z","snapshot_observed_at":"2026-08-07T04:11:37.848436Z","submitted_at":"2026-04-05T21:08:00Z","title":"Data-Driven Boundary Control of Distributed Port-Hamiltonian Systems","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-13T10:21:59.840828Z"},"links":{"cited_paper":"/paper/2602.05843","citing_paper":"/paper/2604.04266"},"observation_digest":"sha256:28e70679cc9c6f17d73b0d9c59cf2e16074b21b22d3340e876a9e893db35e773","observation_id":"1605ec29-59e7-4226-a88a-3b8ec39d37e8","resolution":{"observed_at":"2026-07-13T10:21:59.840828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"cited_work":{"arxiv_id":"2602.05843","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.05843","snapshot_observed_at":"2026-07-04T13:29:51.596096Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","venue":"cs.CL","work_id":"5c625304-8195-497b-9482-04ec8957febc","year":2026},"citing_paper":{"arxiv_id":"2606.26790","last_updated":"2026-06-25T09:24:09Z","snapshot_observed_at":"2026-08-05T22:45:03.217186Z","submitted_at":"2026-06-25T09:24:09Z","title":"OPID: On-Policy Skill Distillation for Agentic Reinforcement Learning","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-26T05:07:16.926615Z"},"links":{"cited_paper":"/paper/2602.05843","citing_paper":"/paper/2606.26790"},"observation_digest":"sha256:431481ac3408f56c4ec7c5a99b760c5727bbf943485d3b4946374f5541962689","observation_id":"21c0901f-91b2-4d8e-9d21-c68fd5fcdc63","resolution":{"observed_at":"2026-07-04T13:29:51.597617Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.05843","snapshot_observed_at":"2026-08-02T01:10:21.243699Z","title":"arXiv preprint arXiv:2602.05843 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.14777","last_updated":"2026-07-16T09:57:18Z","snapshot_observed_at":"2026-08-05T21:32:12.968039Z","submitted_at":"2026-07-16T09:57:18Z","title":"SEED: Self-Evolving On-Policy Distillation for Agentic Reinforcement Learning","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-02T01:10:21.243699Z"},"links":{"cited_paper":"/paper/2602.05843","citing_paper":"/paper/2607.14777"},"observation_digest":"sha256:4a63d45f43ff3ede73c9a7a7ec43a027d240f592640e7f13f3aa8a235e0a031e","observation_id":"af3179df-4eb4-4276-ac8f-2bcd261284e6","resolution":{"observed_at":"2026-08-02T01:10:21.243699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.05843","snapshot_observed_at":"2026-07-31T02:18:01.888233Z","title":"Odysseyarena: Benchmarking large language models for long-horizon, active and inductive interactions.arXiv preprint arXiv:2602.05843, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.28609","last_updated":"2026-08-06T17:55:59Z","snapshot_observed_at":"2026-08-07T06:40:02.582832Z","submitted_at":"2026-07-30T17:57:41Z","title":"OSReward: Instituting Standardized Evaluation for Cross-Platform Computer-Use Reward Models","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-07-31T02:18:01.888233Z"},"links":{"cited_paper":"/paper/2602.05843","citing_paper":"/paper/2607.28609"},"observation_digest":"sha256:a81e9bc82d43683c4e058d257d2fd73764ce8ed7f5a36b5331d53dab98e35afb","observation_id":"e53c8bbd-d6b2-4c87-a4c7-57212f0f7113","resolution":{"observed_at":"2026-07-31T02:18:01.888233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2602.05843/citation-record","integrity":"/paper/2602.05843/integrity","json":"/paper/2602.05843/citation-record.json","paper":"/paper/2602.05843"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:08.370940Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:08.370940Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:1abd5f739f98d3b19d26e1c6517b660b3a7943a16110fe6923b280b11577e077","observation_id":"abcf9bb2-67e3-4b98-a511-9cdc52c13b92","resolution":{"observed_at":"2026-08-03T04:08:08.370940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:08.453393Z","title":"J., Bethge, M., and Schulz, E","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:08.453393Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:4aff112e96897eddc19a536dda44e5d375dabf2e04041e43ba71694cdd142785","observation_id":"07be5ed1-70db-4ffe-b7a9-17b09f240e50","resolution":{"observed_at":"2026-08-03T04:08:08.453393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:08.572461Z","title":"The claude 3 model family: Opus, sonnet, haiku","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:08.572461Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:ef282b3255acab88f2df942ad4ae42e4f4b75962a3d5dd9b01b1b0b78457500d","observation_id":"94800fc7-0706-4bf2-8fb4-90643d4fc806","resolution":{"observed_at":"2026-08-03T04:08:08.572461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:08.670628Z","title":"H., and Bengio, Y","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:08.670628Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:65fcf01f5170c57dc4c85a73e3c9aebf70d9ab01d25c5df5894a56541b75c10c","observation_id":"57ffcf1c-e6a2-4347-b8e3-5696591186d4","resolution":{"observed_at":"2026-08-03T04:08:08.670628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1911.01547","last_updated":"2019-11-25T13:02:04Z","snapshot_observed_at":"2026-07-06T08:34:41.399203Z","submitted_at":"2019-11-05T00:31:38Z","title":"On the Measure of Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.01547","snapshot_observed_at":"2026-08-03T04:08:08.735295Z","title":"On the measure of intelligence","venue":null,"work_id":null,"year":1911},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:08.735295Z"},"links":{"cited_paper":"/paper/1911.01547","citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:5fb4ef16bce6687e5128fafc182b9eeb2e68df0bffb56f2d57be29a360930b3c","observation_id":"16f03f94-b2e8-4fb0-b521-7d6d0098e075","resolution":{"observed_at":"2026-08-03T04:08:08.735295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:08.828452Z","title":"Evaluating long-context reasoning in llm-based webagents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:08.828452Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:0d6e544ce1858149cda038a1ed609d66c75b7798fa4a1d9d6008555d75f3ae74","observation_id":"3fe0ebb9-7175-4866-a10c-225c9e067939","resolution":{"observed_at":"2026-08-03T04:08:08.828452Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:09.003347Z","title":"The dynamical challenge","venue":null,"work_id":null,"year":1997},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:09.003347Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:e0cc8cbf183acb534e9c616709f11b3eb8b4d354e36e67afed05200c88ef6e6c","observation_id":"be7fd5c4-2ec1-406e-a03b-57bd71bd983c","resolution":{"observed_at":"2026-08-03T04:08:09.003347Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:09.106776Z","title":"Mind2web: Towards a generalist agent for the web","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:09.106776Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:06c470ece85732ab96b9ef9260717190ae092211939131cb21309388d206618f","observation_id":"62382c70-fc00-4411-b000-1b900451b63a","resolution":{"observed_at":"2026-08-03T04:08:09.106776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-08-03T04:08:09.262712Z","title":"Gemini 2.5: Pushing the frontier with advanced reasoning, multimodality, long context, and next generation agentic capabilities","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:09.262712Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:0afb343968184976a946de4b5bd53c98e944c5ece4648f3affd178d9f9057953","observation_id":"44797934-6322-4048-b2ff-eb7641e983e9","resolution":{"observed_at":"2026-08-03T04:08:09.262712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12793","snapshot_observed_at":"2026-08-03T04:08:09.404493Z","title":"Chatglm: A family of large language models from glm-130b to glm-4 all tools","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:09.404493Z"},"links":{"cited_paper":"/paper/2406.12793","citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:e87a3e929bad63c258e767befa19d1108df8408154e2ddec1cfb93aa2fdd79d2","observation_id":"af6b907a-38c2-4a0f-be18-bc1e33218f2f","resolution":{"observed_at":"2026-08-03T04:08:09.404493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-03T04:08:09.547578Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:09.547578Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:c5a33c92d4c3664605c093d2a7c505b224c5383f722fa759562d20e19eda0c80","observation_id":"1e10462f-7d50-4565-9820-223e71e219ff","resolution":{"observed_at":"2026-08-03T04:08:09.547578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:09.716497Z","title":"and Schmidhuber, J","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:09.716497Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:c6943de7f050039ebce8e240a50a6b5689fb03ececc187cecc363975b3770cf2","observation_id":"ed61fcd1-4de9-4f12-b221-08dfed5f1429","resolution":{"observed_at":"2026-08-03T04:08:09.716497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:09.905299Z","title":"H., Gonzalez, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:09.905299Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:053f00274616bc790ad80363befaf58ce70811174a31c0ce0746e4b3536da450","observation_id":"9cec0ab3-7747-4553-af99-c627e6a1a99b","resolution":{"observed_at":"2026-08-03T04:08:09.905299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:10.115144Z","title":"M., Ullman, T","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:10.115144Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:0588c877dea31a82c6511a17dd2c17005fe2573aa956cb5bf95cd3f990cc5e67","observation_id":"8ff668df-7ab4-408c-8124-43c317ff134d","resolution":{"observed_at":"2026-08-03T04:08:10.115144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:10.290501Z","title":"State space models on temporal graphs: A first-principles study","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:10.290501Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:0ef948d70c99d55daa4be83ae9b3e4d590a0d463f3747955ccb60a7ef2d18d6b","observation_id":"5ac8cd97-6d31-4f57-9d06-5cf1245b2007","resolution":{"observed_at":"2026-08-03T04:08:10.290501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:10.419920Z","title":"Y., Le Bras, R., Richardson, K., Sabharwal, A., Poovendran, R., Clark, P., and Choi, Y","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:10.419920Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:8ba78e0f02a4bbcd633c302b25f3727eff564893c6b08a42669b35dd00eb7c88","observation_id":"2cd6a435-f4ba-42ab-b273-ca7db33800c4","resolution":{"observed_at":"2026-08-03T04:08:10.419920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2512.02556","last_updated":"2025-12-02T09:25:14Z","snapshot_observed_at":"2026-07-31T23:49:25.878472Z","submitted_at":"2025-12-02T09:25:14Z","title":"DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.02556","snapshot_observed_at":"2026-08-03T04:08:10.546967Z","title":"Deepseek-v3","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:10.546967Z"},"links":{"cited_paper":"/paper/2512.02556","citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:2a677045c1da8ca59a1ca0b7704a92f77dc9ec172149e0793552c71862720688","observation_id":"1852803b-9dba-413b-b13f-e8e029d63b47","resolution":{"observed_at":"2026-08-03T04:08:10.546967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:10.676369Z","title":"Agentbench: Evaluating llms as agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:10.676369Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:3540e29dd2a2848e8126373c80aaf03887a19ce7437d0b95d3ce327b4079446b","observation_id":"c604330b-a965-4290-9c52-a8cef996ec79","resolution":{"observed_at":"2026-08-03T04:08:10.676369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:10.801903Z","title":"Gaia: a benchmark for general ai assistants","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:10.801903Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:6c83a3caa04db0e29c7c5f310a913eda6d16304b5fbf55a4ddc66d90e9c77eb4","observation_id":"dfa35dbd-6872-4218-ac85-b64b4971485e","resolution":{"observed_at":"2026-08-03T04:08:10.801903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10925","last_updated":"2025-08-08T19:24:38Z","snapshot_observed_at":"2026-08-01T16:27:35.664983Z","submitted_at":"2025-08-08T19:24:38Z","title":"gpt-oss-120b & gpt-oss-20b Model Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.10925","snapshot_observed_at":"2026-08-03T04:08:10.909307Z","title":"gpt-oss-120b & gpt-oss-20b model card","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:10.909307Z"},"links":{"cited_paper":"/paper/2508.10925","citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:3812780f13a4517c1b086df820c22a5d61fd2f42cdab73527885845d7d4b3ccd","observation_id":"85a10ad0-493f-4700-91ed-7e7dd255d1c3","resolution":{"observed_at":"2026-08-03T04:08:10.909307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:11.042144Z","title":"G., Mao, H., Yan, F., Ji, C","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:11.042144Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:1341a2812bfac1a571bda54d739ec1c544a285991f9513b47fbb46647cca92f6","observation_id":"23bf426b-6ed8-4bef-ac19-2290650ab49b","resolution":{"observed_at":"2026-08-03T04:08:11.042144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:11.173177Z","title":"E., Li, W., Campbell-Ajala, F., Toyama, D","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:11.173177Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:404170f937ed829e6030783a37112f0215e6edf5a26f3c7653d5e23e59445f52","observation_id":"b09a2020-4b4a-47e9-954c-030b1dd6c2b7","resolution":{"observed_at":"2026-08-03T04:08:11.173177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:11.342942Z","title":"Reflexion: Language agents with verbal reinforcement learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:11.342942Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:cc9cfae6177e35cef6156ac8ac544284fd6b0a1eaa04cec90a9bba6c0d41d53f","observation_id":"0a45872a-b04e-43aa-ac18-15242df16764","resolution":{"observed_at":"2026-08-03T04:08:11.342942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:11.473340Z","title":"Alfworld: Aligning text and embodied environments for interactive learning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:11.473340Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:0d58dc9e40da60015b1e17fa394da093258c78ca6979eb793e438275e7b6107b","observation_id":"a6e9b9cd-6bbb-496e-b748-f03955026ff5","resolution":{"observed_at":"2026-08-03T04:08:11.473340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00280","last_updated":"2024-08-21T05:11:10Z","snapshot_observed_at":"2026-07-06T16:25:50.576054Z","submitted_at":"2023-09-30T07:11:39Z","title":"Corex: Pushing the Boundaries of Complex Reasoning through Multi-Model Collaboration","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00280","snapshot_observed_at":"2026-08-03T04:08:11.637834Z","title":"Corex: Pushing the boundaries of complex reasoning through multi-model collaboration","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:11.637834Z"},"links":{"cited_paper":"/paper/2310.00280","citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:eb9f76bec436551d0bbc2cedb73ee05782cba068f6986497ec71e23c4d60f45e","observation_id":"631d89fc-a2a0-4684-9021-44f972d26f30","resolution":{"observed_at":"2026-08-03T04:08:11.637834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:11.767750Z","title":"Os-genesis: Automating gui agent trajectory construction via reverse task synthesis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:11.767750Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:dbf21b04f0d6cd879305e805436a25080537934770fed928e79e83b6b3ffab87","observation_id":"0ba0fedd-aab4-41ee-9055-510c069421ca","resolution":{"observed_at":"2026-08-03T04:08:11.767750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19897","last_updated":"2026-04-20T16:29:04Z","snapshot_observed_at":"2026-07-06T21:30:37.657153Z","submitted_at":"2025-05-26T12:27:27Z","title":"ScienceBoard: Evaluating Multimodal Autonomous Agents in Realistic Scientific Workflows","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19897","snapshot_observed_at":"2026-08-03T04:08:11.873882Z","title":"Scienceboard: Evaluating multimodal autonomous agents in realistic scientific workflows","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:11.873882Z"},"links":{"cited_paper":"/paper/2505.19897","citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:4a1208d6d7df817d048698d55f8e4523366a7b14ae1fa60354d56540e209a1c4","observation_id":"b568cec5-36ea-4275-8471-e1aaa957c2bd","resolution":{"observed_at":"2026-08-03T04:08:11.873882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:12.006468Z","title":"Mars: Situated inductive reasoning in an open-world environment","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:12.006468Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:c5fd9ccc5d646b2757c88b3761b89c9842898a31b34ca04c944093c7f74e7103","observation_id":"5457a67c-5fef-457e-a935-c1c5ea5a7185","resolution":{"observed_at":"2026-08-03T04:08:12.006468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:12.083678Z","title":"Michelangelo: Long context evaluations beyond haystacks via latent structure queries","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:12.083678Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:4e759c83ea633ddbaf6857315df0c40540367352c2996585f7623689175e6c61","observation_id":"5db826d9-a19c-4df5-8b05-6c831e77eb1d","resolution":{"observed_at":"2026-08-03T04:08:12.083678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:12.161041Z","title":"Large language models for robotics: Opportunities, challenges, and perspectives","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:12.161041Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:d4b5cb4c0a5cb81d5d0ad038a2669cc62874f0e415bd392fc3d8a87cb8ed6c3e","observation_id":"b861cdd6-77b0-4f70-816b-14f279c5dd1e","resolution":{"observed_at":"2026-08-03T04:08:12.161041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.09124","last_updated":"2025-08-12T17:53:03Z","snapshot_observed_at":"2026-08-06T17:04:47.204008Z","submitted_at":"2025-08-12T17:53:03Z","title":"OdysseyBench: Evaluating LLM Agents on Long-Horizon Complex Office Application Workflows","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.09124","snapshot_observed_at":"2026-08-03T04:08:12.299526Z","title":"M., Xu, J., R \\\"u hle, V., and Rajmohan, S","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:12.299526Z"},"links":{"cited_paper":"/paper/2508.09124","citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:fa570fe91f7b751ab533427aef8af812515e2261aaebb3f25099cd90f54d9a8e","observation_id":"5072d6c0-9b4a-4c27-8feb-cd934da8508e","resolution":{"observed_at":"2026-08-03T04:08:12.299526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12516","last_updated":"2025-04-16T22:27:45Z","snapshot_observed_at":"2026-08-03T00:43:33.338074Z","submitted_at":"2025-04-16T22:27:45Z","title":"BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.12516","snapshot_observed_at":"2026-08-03T04:08:12.409116Z","title":"W., Passos, A","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:12.409116Z"},"links":{"cited_paper":"/paper/2504.12516","citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:bc6656586e6ecc00f6835ef37c187e74738cebb59b56df2fdd559066b33fe430","observation_id":"7dc572a8-0df3-451e-b9ac-a00441bee7b8","resolution":{"observed_at":"2026-08-03T04:08:12.409116Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:12.589507Z","title":"J., Cheng, Z., Shin, D., Lei, F., et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:12.589507Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:2a0246b8405c82bc4f7fdd6a46b0be739d8125f4cd21ffa96ccb3254679a9983","observation_id":"3d64ee9c-600e-4045-91ef-3e82ee3fccb2","resolution":{"observed_at":"2026-08-03T04:08:12.589507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:12.822243Z","title":"-decoding: Adaptive foresight sampling for balanced inference-time exploration and exploitation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:12.822243Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:3e245c9264c23244020b8a93d8290b5fb487aaa718d26e63b391dda8e0f3f006","observation_id":"44a7aa4d-0634-4392-a1d1-38bc028e9b73","resolution":{"observed_at":"2026-08-03T04:08:12.822243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.acl-long.644","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Genius: A generalizable and purely unsupervised self-training framework for advanced reasoning","venue":null,"work_id":"793985c7-a465-4667-a10e-8252899ac35e","year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:13.072792Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:67e0c12cf8dda30adcca50574e47734eb917f14bbfebcb05cece1b24a5587a9b","observation_id":"8845d5c2-7533-412e-9d91-966758b90c40","resolution":{"observed_at":"2026-08-03T04:08:22.679861Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:13.225646Z","title":"F., Song, Y., Li, B., Tang, Y., Jain, K., Bao, M., Wang, Z","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:13.225646Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:b9134f54fbc7edc992b362f8656bb2d21c1c878bb3fb7b8550243746394b439c","observation_id":"75d7f364-33cf-4678-b519-71aa044782ed","resolution":{"observed_at":"2026-08-03T04:08:13.225646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:13.333287Z","title":"Tide: Trajectory-based diagnostic evaluation of test-time improvement in llm agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:13.333287Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:ee1a1389f27e625f88898f0473b6d06e237668c7176b340c4c91acf857553b3f","observation_id":"dd8d17e4-1205-4db8-a761-618e5d003638","resolution":{"observed_at":"2026-08-03T04:08:13.333287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-03T04:08:13.491585Z","title":"Qwen3 technical report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:13.491585Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:7b6d88daf5c092f61a6d0b5146c80e929ad70e79ffea7de4fb09381f4d706173","observation_id":"2e1afbbe-60b3-4e4a-b98c-8b4a7d61c8ff","resolution":{"observed_at":"2026-08-03T04:08:13.491585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:13.702944Z","title":"R., and Cao, Y","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:13.702944Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:9f571b0b6c9371e9371851a0cf21bf39c6d7e618b3949e0fdab55b196b31338e","observation_id":"a70c3329-8764-4e79-8ab6-3bf806c448e3","resolution":{"observed_at":"2026-08-03T04:08:13.702944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:13.920742Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:13.920742Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:f3d7f16ca1bce93626269b71ecfec0c54527d49c75c29bb5fdbff137decfa75a","observation_id":"9fcf34c6-6b1a-4498-9210-432e801688f0","resolution":{"observed_at":"2026-08-03T04:08:13.920742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:08:14.068241Z","title":"F., Zhu, H., Zhou, X., Lo, R., Sridhar, A., Cheng, X., Ou, T., Bisk, Y., Fried, D., et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-03T04:08:14.068241Z"},"links":{"citing_paper":"/paper/2602.05843"},"observation_digest":"sha256:42772366517f8df96fbfd1eda0e7436a1c31167766992a4ed3ec4575cfbde0a1","observation_id":"610b432b-f0ae-41c2-9860-b129f41b0f91","resolution":{"observed_at":"2026-08-03T04:08:14.068241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2602.05843","last_updated":"2026-06-04T17:59:52Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T01:22:03.673149Z","submitted_at":"2026-02-05T16:31:43Z","title":"OdysseyArena: Benchmarking Large Language Models For Long-Horizon, Active and Inductive Interactions"},"reference_resolution":{"displayed":41,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":40,"verified_exact":1,"verified_fuzzy":0},"total_outbound_references":41},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 41 of 41 outbound references and 4 inbound Pith citation observations for arXiv:2602.05843."}