{"as_of":"2026-08-09T14:16:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2bb13a941f6953c16078be451f8f37053f51ab36257b80d86fb54c014881a457","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T23:30:43.662827Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":6,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:26:56.639836Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-25T08:15:33.926301Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":"2502.08859","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enigmaeval: A benchmark of long multimodal reasoning challenges","venue":null,"work_id":"0cc9ff97-d72f-4232-b771-d3c969ededbc","year":2025},"citing_paper":{"arxiv_id":"2504.19678","last_updated":"2026-03-06T19:01:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-28T11:08:22Z","title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-15T02:57:37.873567Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2504.19678"},"observation_digest":"sha256:0c563630999a3b0189c11b66032e0a946e6f222374467a3e72c535de1094d918","observation_id":"94a3833b-a525-47d0-b967-5d248b0f689d","resolution":{"observed_at":"2026-05-15T02:57:38.309833Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-08-07T15:26:56.639836Z","title":"arXiv preprint arXiv:2502.08859 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-07T21:22:21.405277Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.639836Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:4ab189391cd68da460f7750d3ec0e38c37c95531fc37f69cd91e8afdaef42cff","observation_id":"57530e12-cda5-41fe-bf7f-6574601f213e","resolution":{"observed_at":"2026-08-07T15:26:56.639836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-08-07T15:08:30.334466Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.16135","last_updated":"2025-05-22T02:24:35Z","snapshot_observed_at":"2026-08-09T09:48:39.463552Z","submitted_at":"2025-05-22T02:24:35Z","title":"Sudoku-Bench: Evaluating creative reasoning with Sudoku variants","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:30.334466Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2505.16135"},"observation_digest":"sha256:be1357345ff4ac276e41630d548413beac344750907d8fd7e5be52413c780ebc","observation_id":"e80c223a-4010-4c5b-ac75-717ba7a021e7","resolution":{"observed_at":"2026-08-07T15:08:30.334466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":"2502.08859","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enigmaeval: A benchmark of long multimodal reasoning challenges","venue":null,"work_id":"0cc9ff97-d72f-4232-b771-d3c969ededbc","year":2025},"citing_paper":{"arxiv_id":"2506.06211","last_updated":"2026-04-21T03:40:06Z","snapshot_observed_at":"2026-08-03T01:28:16.465454Z","submitted_at":"2025-06-06T16:17:09Z","title":"PuzzleWorld: A Benchmark for Multimodal, Open-Ended Reasoning in Puzzlehunts","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-19T10:43:02.601014Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2506.06211"},"observation_digest":"sha256:19a4b51e65e6d0588b9e2d927246ef96872ad8c1abf160da3a9c01ef0542f127","observation_id":"3b4b129e-89df-44b5-ae33-826b8ebefe91","resolution":{"observed_at":"2026-05-19T10:47:15.164738Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":"2502.08859","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enigmaeval: A benchmark of long multimodal reasoning challenges","venue":null,"work_id":"0cc9ff97-d72f-4232-b771-d3c969ededbc","year":2025},"citing_paper":{"arxiv_id":"2510.08945","last_updated":"2026-05-21T22:37:46Z","snapshot_observed_at":"2026-07-06T22:32:17.816175Z","submitted_at":"2025-10-10T02:51:47Z","title":"FATHOMS-RAG: A Framework for the Assessment of Thinking and Observation in Multimodal Systems that use Retrieval Augmented Generation","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-25T08:13:05.328746Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2510.08945"},"observation_digest":"sha256:147fab239a55d7e0945912f518fa261d6470288ceecb8ccfbb326c52b53289df","observation_id":"048dd1b5-73d0-4c8c-aeca-b028b91d41fe","resolution":{"observed_at":"2026-05-25T08:15:33.930689Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":"2502.08859","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enigmaeval: A benchmark of long multimodal reasoning challenges","venue":null,"work_id":"0cc9ff97-d72f-4232-b771-d3c969ededbc","year":2025},"citing_paper":{"arxiv_id":"2605.14040","last_updated":"2026-05-13T19:00:57Z","snapshot_observed_at":"2026-07-06T23:25:29.856145Z","submitted_at":"2026-05-13T19:00:57Z","title":"Physics-R1: An Audited Olympiad Corpus and Recipe for Visual Physics Reasoning","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-15T05:35:32.806871Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2605.14040"},"observation_digest":"sha256:f4316f264b491ec6d69b793974e30e2b270be9c7bc0dcf6795eba5db797fcfc5","observation_id":"8ee27dfd-b016-448f-9762-eac45709a9c7","resolution":{"observed_at":"2026-05-15T05:39:47.813355Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2502.08859/citation-record","integrity":"/paper/2502.08859/integrity","json":"/paper/2502.08859/citation-record.json","paper":"/paper/2502.08859"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.152706Z","title":"https://puzzledpint.org/","venue":null,"work_id":"e197f0bc-b382-4b43-b193-d33bfe557cd6","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.527786Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:2ddeef3227fcb9134eb2234b5c75994c30573f454bccca355a278ba1875ca86c","observation_id":"82346c5b-96d1-42e8-81cb-8cd0681ce094","resolution":{"observed_at":"2026-08-07T23:30:44.156381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.141905Z","title":null,"venue":null,"work_id":"5ef93580-5187-43c3-8366-df8dfb72cc8e","year":2025},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.531958Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:4f06b417c83000046943cc2882b501933af4194480e05bb982bc939fe5dcd764","observation_id":"843cd5f7-0e66-41bd-a6cc-633ae07e1a93","resolution":{"observed_at":"2026-08-07T23:30:44.145614Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.130060Z","title":"Puzzle Potluck","venue":null,"work_id":"d52b91ff-5428-440a-b23b-c3023dc7a109","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.535582Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:4553eac5f0e1689c942bdeda962732944356139b7522011b5c256574fbaa9afb","observation_id":"5c814eac-f57c-4476-a4be-ec1b1dcaaf93","resolution":{"observed_at":"2026-08-07T23:30:44.133987Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.119456Z","title":"SOME PUZZLES by Mark Halpin","venue":null,"work_id":"1b9fe7f0-4cf0-4511-9162-b6675e8597dc","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.539340Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:a89b3af0c752626281879373cd8a9c895293fd120da7c5d3e189f292a79c8327","observation_id":"07555f79-d9b5-4272-9dae-060dc3c4e83d","resolution":{"observed_at":"2026-08-07T23:30:44.123189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.107955Z","title":null,"venue":null,"work_id":"05af4f2e-6ad0-4550-b702-17293f96981b","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.542911Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:c96f13029b348bc72b9604c6fe78d541fd0fdc4b527112b8cd299c9e42709b4b","observation_id":"5126262f-5fa0-42a8-875d-6c8ad29cd47c","resolution":{"observed_at":"2026-08-07T23:30:44.112500Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.097280Z","title":"https://puzzles.mit.edu/","venue":null,"work_id":"5e321ef2-0ed6-42a0-95e6-d0292e1a7aa2","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.546828Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:c5c8c83b08bd5586681a811c2dbf261ef8c7173d5ee8a83356ff9d7845736a72","observation_id":"8ae17add-ffd2-4fa9-b4b2-a6c2515e9957","resolution":{"observed_at":"2026-08-07T23:30:44.100847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.085985Z","title":"Grandmaster Puzzles","venue":null,"work_id":"7d4d5a31-906b-4952-8f8f-e87c59499702","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.550584Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:455398aab0f7517f460fac4735643bb02bc23210e65a517c20361c68575c9b9e","observation_id":"b2dfcc65-8744-4ff9-81e1-10de40e61f08","resolution":{"observed_at":"2026-08-07T23:30:44.090026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.554415Z","title":"Measuring mathematical problem solving with the math dataset, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.554415Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:8b7f6e0f388231495fbaf8986830b306aab76722a6348e1acb7aff6ec247c5be","observation_id":"7cd582dc-3fd8-469c-a10f-7f7fbcb531af","resolution":{"observed_at":"2026-08-07T23:30:43.554415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.557938Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.557938Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:e841fc95d511692e0ea03384ce8afd63025dfb90546fd8b737bdbcecb82f2905","observation_id":"391adf69-e187-44ba-acb9-c27e1c67651b","resolution":{"observed_at":"2026-08-07T23:30:43.557938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.060792Z","title":"Frontiermath: A benchmark for evaluating advanced mathematical reasoning in ai, 2024","venue":null,"work_id":"b1274e2b-dd59-4f59-96d2-a2c4fe784473","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.561371Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:44e17557e987032882f3bd2ffcdb5a2ee599e87368733a9f9914148714dec285","observation_id":"9d881114-a65c-439d-850e-23a6e4529837","resolution":{"observed_at":"2026-08-07T23:30:44.065829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.564791Z","title":"Olympiadbench: A challenging benchmark for promoting agi with olympiad-level bilingual multimodal scientific problems, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.564791Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:3a6978b94c4446101a82997cb69bde78af5ef619be4ea7eb91d08b6203900c1e","observation_id":"0cd4cd2d-5df6-4280-bfbb-f1383fef72a4","resolution":{"observed_at":"2026-08-07T23:30:43.564791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.043119Z","title":"Humanity’s Last Exam, 2025","venue":null,"work_id":"c7086602-f360-4727-bd77-73b952aa0471","year":2025},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.568159Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:c6224d9b07c32028273ee2c084fa72d16b08b70926f549c4e57715eb28d6d405","observation_id":"7d64a4a5-0732-4d6c-bbf7-a631a43aa849","resolution":{"observed_at":"2026-08-07T23:30:44.046677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.571998Z","title":"Measuring massive multitask language understanding, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.571998Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:f71c9ccef1d5fd590ced1de3a72ca04d14bd753f8357effe4201ee58b6bffd62","observation_id":"b177cc5e-d96f-4155-9f66-ae21f995b0b1","resolution":{"observed_at":"2026-08-07T23:30:43.571998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.026335Z","title":"Mmmu: A massive multi- discipline multimodal understanding and reasoning benchmark for expert agi","venue":null,"work_id":"6eb46864-850d-467f-b9bb-212df0b91540","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.575607Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:b75b123a92016457bf1cca103d1ed561a2ffd72411bf67ed4d21734442a59f1e","observation_id":"8f5d7d3e-4517-4098-8eb6-4c80358f5981","resolution":{"observed_at":"2026-08-07T23:30:44.030082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.578906Z","title":"Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.578906Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:8563f38f14869a53de0266de186398f9eebc106809a28b421413fd474b51ad96","observation_id":"65f7a8ab-ff2c-48cc-a61b-b9bc36381e13","resolution":{"observed_at":"2026-08-07T23:30:43.578906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.010171Z","title":"Vista: A rubric- based visual task assessment","venue":null,"work_id":"35be598a-a734-40b3-b803-84337149d259","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.582228Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:51b85be193f7daf59d50c1e5ec8750c6e7ddb09e4b0b2c2468328f15ec5a1a7a","observation_id":"460bbc65-3f94-4551-9fc9-f93c875e9445","resolution":{"observed_at":"2026-08-07T23:30:44.013743Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.585618Z","title":"On the measure of intelligence, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.585618Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:3bf0a6a58a5dbb5e9ac77b4804193fa19b43a0c78d0c3b6cf9a58943a38dfd89","observation_id":"cfc927e1-817f-462c-a866-d9bbabc01ebe","resolution":{"observed_at":"2026-08-07T23:30:43.585618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.993984Z","title":"Lanzendörfer, Yannick Niedermayr, and Roger Wattenhofer","venue":null,"work_id":"b7f7aaaa-f22a-46ed-8af2-3d36fd5e9019","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.588907Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:4c2a489a4727cdc82e5942246c2faa9613c9bbecacc8839c24f2c0dbc8be9155","observation_id":"8c0601a4-d84c-47d8-905a-eef165df06a6","resolution":{"observed_at":"2026-08-07T23:30:43.997755Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.983687Z","title":"PuzzlePlex: A Benchmark to Evaluate the Reasoning and Planning of Large Language Models on Puzzles, 2025","venue":null,"work_id":"9d6e9343-2f93-47d2-a074-0ce4d5134811","year":2025},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.592101Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:e00efcefadc914a9320a009e5f90814aa1ba7f0107825b325041f46c9623b9af","observation_id":"5ac52958-4155-4a5f-98e0-eb7b092d71c4","resolution":{"observed_at":"2026-08-07T23:30:43.987521Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.02611","last_updated":"2025-03-01T12:46:25Z","snapshot_observed_at":"2026-08-06T18:34:26.106835Z","submitted_at":"2024-02-04T20:56:09Z","title":"FCoReBench: Can Large Language Models Solve Challenging First-Order Combinatorial Reasoning Problems?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.02611","snapshot_observed_at":"2026-08-07T23:30:43.595386Z","title":"PuzzleBench: Can LLMs Solve Challenging First-Order Combinatorial Reasoning Problems? arXiv preprint arXiv:2402.02611, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.595386Z"},"links":{"cited_paper":"/paper/2402.02611","citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:7043c1e5b871c1c6490b0c8059b45e77f08b7dd9114926ae9545362baf8de5d0","observation_id":"d18500c4-f372-48a4-a7e0-2d5e38939151","resolution":{"observed_at":"2026-08-07T23:30:43.595386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.14790","last_updated":"2024-10-04T04:58:12Z","snapshot_observed_at":"2026-07-06T18:49:25.288816Z","submitted_at":"2024-07-20T07:43:07Z","title":"Step-by-Step Reasoning to Solve Grid Puzzles: Where do LLMs Falter?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.14790","snapshot_observed_at":"2026-08-07T23:30:43.599486Z","title":"Step-by-Step Reasoning to Solve Grid Puzzles: Where do LLMs Falter? arXiv preprint arXiv:2407.14790, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.599486Z"},"links":{"cited_paper":"/paper/2407.14790","citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:e821059381265729bf9711aefae513fe53900a73557a932a581911fde054a320","observation_id":"46e6cb43-c7dd-4a51-8eb6-9c7d8dccd28e","resolution":{"observed_at":"2026-08-07T23:30:43.599486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00376","last_updated":"2021-07-04T22:50:32Z","snapshot_observed_at":"2026-08-09T00:04:22.828648Z","submitted_at":"2021-01-02T05:28:15Z","title":"RiddleSense: Reasoning about Riddle Questions Featuring Linguistic Creativity and Commonsense Knowledge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.00376","snapshot_observed_at":"2026-08-07T23:30:43.603305Z","title":"Riddlesense: Answering riddle questions as commonsense reasoning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.603305Z"},"links":{"cited_paper":"/paper/2101.00376","citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:90bec5e913b4edb11a8ff5350b6ff7b3d7d4809c562e71d1ff75a64819d86680","observation_id":"d7a2bf39-b9b4-4d99-a75d-af9f8f300053","resolution":{"observed_at":"2026-08-07T23:30:43.603305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.973643Z","title":"https://www.melbunimathsstats.org/puzzlehunt","venue":null,"work_id":"de9e17a2-e887-4fad-9b54-06fa8281621c","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.607095Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:28ffd29c8ba2a1e083ec68a34f660a68d3208256a8a825d5559e22d6f63b701d","observation_id":"e6bf3606-0828-4f48-9380-73d8c2b2cc1d","resolution":{"observed_at":"2026-08-07T23:30:43.977237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.610435Z","title":"https://web.archive.org/web/20210725192741/https://www","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.610435Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:16feac93b4ecdb2f8919b2ffe43a08a8f63535ee834778a6f83981043f0e4948","observation_id":"b1cee88c-c0d9-446e-b0f4-45668caf367e","resolution":{"observed_at":"2026-08-07T23:30:43.610435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.964008Z","title":"https://harvardpuzzles.github.io/","venue":null,"work_id":"2675e864-cbfb-44d9-b8a6-e49206236817","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.613709Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:4f6202dae4d25dbb3675cfc8223b02f04786654a4ffde1a5c5c7c4dfe895dcbf","observation_id":"342ba2ad-cb5f-49c1-ae7e-e7f7828dfbdd","resolution":{"observed_at":"2026-08-07T23:30:43.967490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.954153Z","title":"https://www.mezzacotta.net/puzzle/cisra/","venue":null,"work_id":"ec4c73ac-dc9f-4fbe-a621-92c76f8bc86f","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.616875Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:45671cdd647a3e2a43c9041419f84e34a860c7fa56e7ca85c74abbafd23251de","observation_id":"c7fdeb96-070a-4dc1-aaeb-f1ba1088d752","resolution":{"observed_at":"2026-08-07T23:30:43.957817Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.944353Z","title":"https://www.janestreet.com/puzzles/archive/index.html","venue":null,"work_id":"11cb608a-1090-47aa-ac3f-aa5d5d178887","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.620473Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:209cf4b16ddc5e90d2a78a196b14df9f80825aa797f85cd7b8115c4c5ae722b6","observation_id":"503937d6-4c14-453e-ac0c-00bc546d01df","resolution":{"observed_at":"2026-08-07T23:30:43.947801Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.934699Z","title":"https://gooooogol.theburninators.org/puzzles/","venue":null,"work_id":"c09a476b-b4bc-4e3e-8fb7-0dc7b1b3065b","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.624023Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:0f34d7b4b393ae22e1fb396721fb223f3437b1bd5a19f38a5e784f55086965c3","observation_id":"3a1b91e4-717d-4d1e-af44-e612c8e06b68","resolution":{"observed_at":"2026-08-07T23:30:43.938099Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.925243Z","title":"https://playdash.org/","venue":null,"work_id":"3a238520-6b4f-4587-b771-ce62644b17cf","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.627592Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:3e40b85b6e95d5d530a97ccb5fa24827da8bb444f0b45fd75d1add7095bf1288","observation_id":"a1c18164-aca6-46c1-821a-3a837d60b995","resolution":{"observed_at":"2026-08-07T23:30:43.928572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.915660Z","title":"https://www.baphl.org/","venue":null,"work_id":"5f0360c9-0815-4280-b281-f44e3a9109d2","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.631361Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:750f54e8e62ea0c02955e734aba3ac81585dcea4a4a74af2c4787aa22f223b42","observation_id":"1776b629-139b-4d48-a687-d23cda9ee5a3","resolution":{"observed_at":"2026-08-07T23:30:43.918925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.905535Z","title":"Forbidden actions","venue":null,"work_id":"a2dc60e9-d78b-420a-9309-9a5d8a7fddf2","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.634785Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:8062ba80867dcd580b8ce12bd99ffa62a2d8ab7540217e7f2ec4d3b2c1358e3f","observation_id":"f509d110-5f11-4786-99f7-72c04cdb207d","resolution":{"observed_at":"2026-08-07T23:30:43.909101Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.895784Z","title":null,"venue":null,"work_id":"4136f01a-cefe-4c6d-92d8-8b3ad089f192","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.638190Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:018d67cffac71c06f31bad0a2aacec8ef43741bc1cae8227616ffdd19be47daa","observation_id":"49d8ad0e-f128-40dc-9fbe-5178ccecdb90","resolution":{"observed_at":"2026-08-07T23:30:43.899181Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.885485Z","title":null,"venue":null,"work_id":"cab37904-8fb0-47d1-b930-67790dc526c1","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.641496Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:7a8196b7fc2f7e53bc6ff31894f0cc82986ea554c2ae2ac3172600c0373d465b","observation_id":"de4a874d-eb72-42cd-87eb-a1f8537c0f43","resolution":{"observed_at":"2026-08-07T23:30:43.889272Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.874627Z","title":null,"venue":null,"work_id":"75c5b606-e16b-4682-8ce7-f7073793e63e","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.645200Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:16d35a547889e0210b10a19678f25553a6614ef52c1d34b6596ac6453bdb5240","observation_id":"79fa53c8-27b7-4cb2-8e7c-6eb38e6962f6","resolution":{"observed_at":"2026-08-07T23:30:43.878286Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.863673Z","title":null,"venue":null,"work_id":"590bb88c-759e-4dda-8dbc-c5f943f4be9d","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.648403Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:b08c3e07314fe309f29c16f194e08796d056753677baa8b84b8ace57c763fbd8","observation_id":"708a5c9a-11ba-41b1-9db5-99094dc2a39c","resolution":{"observed_at":"2026-08-07T23:30:43.867120Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.853003Z","title":"Problem Web Wage","venue":null,"work_id":"225c7e9c-f81b-43ae-bde8-809cc6e20561","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.651991Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:90833f784c7df827e5a29e63267d1ccb3111c9a5e25609ce35d389601cdc0c77","observation_id":"25a673f7-2449-4694-8b86-389c4c22039d","resolution":{"observed_at":"2026-08-07T23:30:43.856528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.842607Z","title":null,"venue":null,"work_id":"8f70936b-389b-4a96-bad9-9c02df12f1cb","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.655828Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:3481fca6f6d22c8c1e46807e8e342e2cf8497efe4cf4463ebcd7c4deacde3349","observation_id":"ea411947-d5fb-4d7c-8130-9d9b09a33023","resolution":{"observed_at":"2026-08-07T23:30:43.846067Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.832352Z","title":null,"venue":null,"work_id":"d3ec59d8-29ba-4f69-86a5-cc4eda7ee839","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.659550Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:c8cb93d0880726a2d89561a9b898d5ec740bcdab41bc05ec2b4ec914e8a963c1","observation_id":"cd0e5dc9-42fa-4b02-8123-e1e335477d5a","resolution":{"observed_at":"2026-08-07T23:30:43.835897Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.819164Z","title":"This structured approach to answer formats allows us to extract answers consistently and reduces ambiguity when comparing model outputs to ground-truth solutions","venue":null,"work_id":"a0f08d6b-3e75-4922-a5ae-3169996cb8ae","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.662827Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:1d910de5f231cfa629fa08b868e187eb0ff1535f101b14da699e21b2f550a08e","observation_id":"d052f01f-dada-4f97-b0b3-42046345fa91","resolution":{"observed_at":"2026-08-07T23:30:43.824654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":18,"verified_exact":0,"verified_fuzzy":21},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 6 inbound Pith citation observations for arXiv:2502.08859."}