{"as_of":"2026-08-17T05:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4022df01643cda2869f6efebc0f262b49c8c69660fc9112bb5a1e21837fa858f","coverage":[{"denominator":69,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":69,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:35:31.807793Z","state":"measured"},{"denominator":74,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":74,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T00:41:01.511851Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-09T14:26:20.339663Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.10541","snapshot_observed_at":"2026-08-06T14:53:04.542163Z","title":"Vicky Zhao, Conghui He, and Lijun Wu","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17512","last_updated":"2025-07-23T13:51:04Z","snapshot_observed_at":"2026-08-09T23:40:29.433863Z","submitted_at":"2025-07-23T13:51:04Z","title":"Can One Domain Help Others? A Data-Centric Study on Multi-Domain Reasoning via Reinforcement Learning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T14:53:04.542163Z"},"links":{"cited_paper":"/paper/2507.10541","citing_paper":"/paper/2507.17512"},"observation_digest":"sha256:2d8dd5f8f4592eb6966c732dfb301be6bc73c553385b447a08c3f6c96c0e8c42","observation_id":"b4698c59-ecbb-4409-b14c-54f31f80b53c","resolution":{"observed_at":"2026-08-06T14:53:04.542163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.10541","snapshot_observed_at":"2026-08-03T05:43:54.185654Z","title":"V ., He, C., and Wu, L","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.01472","last_updated":"2026-06-24T08:53:47Z","snapshot_observed_at":"2026-08-03T05:43:51.520535Z","submitted_at":"2026-02-01T22:31:19Z","title":"ConPress: Learning Efficient Reasoning from Multi-Question Contextual Pressure","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-03T05:43:54.185654Z"},"links":{"cited_paper":"/paper/2507.10541","citing_paper":"/paper/2602.01472"},"observation_digest":"sha256:89edebebeb8ca8be44e2b180850c7348812437167d747daee7bec61f0ff68ece","observation_id":"c7a683dd-5643-4caf-aeaf-4f7f986a0c69","resolution":{"observed_at":"2026-08-03T05:43:54.185654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"cited_work":{"arxiv_id":"2507.10541","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10541","snapshot_observed_at":"2026-07-09T14:26:20.339663Z","title":"Vicky Zhao, Conghui He, and Lijun Wu","venue":"cs.CL","work_id":"5e714524-d2ac-4fe5-8003-ce63fa95e097","year":2025},"citing_paper":{"arxiv_id":"2604.10480","last_updated":"2026-04-12T06:24:07Z","snapshot_observed_at":"2026-08-14T13:09:40.996128Z","submitted_at":"2026-04-12T06:24:07Z","title":"Tracing the Roots: A Multi-Agent Framework for Uncovering Data Lineage in Post-Training LLMs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T16:14:38.371700Z"},"links":{"cited_paper":"/paper/2507.10541","citing_paper":"/paper/2604.10480"},"observation_digest":"sha256:5c35a0782bfc4a6e28be10fe3d7786231581ce565ce4c795817ae31f5362b3f8","observation_id":"809f36c3-16c5-4fd6-8885-1d3af49c5462","resolution":{"observed_at":"2026-05-11T09:06:00.571138Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"cited_work":{"arxiv_id":"2507.10541","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10541","snapshot_observed_at":"2026-07-09T14:26:20.339663Z","title":"Vicky Zhao, Conghui He, and Lijun Wu","venue":"cs.CL","work_id":"5e714524-d2ac-4fe5-8003-ce63fa95e097","year":2025},"citing_paper":{"arxiv_id":"2607.07321","last_updated":"2026-07-08T12:09:49Z","snapshot_observed_at":"2026-08-12T18:00:34.257614Z","submitted_at":"2026-07-08T12:09:49Z","title":"From Atomic Actions to Standard Operating Procedures: Iterative Tool Optimization for Self-Evolving LLM Agents","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-07-09T14:16:17.970333Z"},"links":{"cited_paper":"/paper/2507.10541","citing_paper":"/paper/2607.07321"},"observation_digest":"sha256:b488ad021227f23e0605ecf8599fa449f6bddc6544e051c4fee17b396422c133","observation_id":"33210c55-ec7d-49c0-8eb3-8996eac8a493","resolution":{"observed_at":"2026-07-09T14:26:20.341497Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.10541","snapshot_observed_at":"2026-08-12T00:41:01.511851Z","title":"Vicky Zhao, Conghui He, and Lijun Wu","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.07968","last_updated":"2026-08-11T02:23:08Z","snapshot_observed_at":"2026-08-14T23:09:45.697398Z","submitted_at":"2026-08-08T07:04:28Z","title":"Thinking Hard, Not Smart: Reasoning Models Fail to Ration Test-Time Compute Across Questions","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-12T00:41:01.511851Z"},"links":{"cited_paper":"/paper/2507.10541","citing_paper":"/paper/2608.07968"},"observation_digest":"sha256:0394c5f54ad26a89239040c919a954f7be67d44fecd52a7fd7da9b19fa61c5f7","observation_id":"c6db5c40-fc03-4847-ac89-107798b2046b","resolution":{"observed_at":"2026-08-12T00:41:01.511851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.10541/citation-record","integrity":"/paper/2507.10541/integrity","json":"/paper/2507.10541/citation-record.json","paper":"/paper/2507.10541"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.04697","last_updated":"2025-10-03T01:55:58Z","snapshot_observed_at":"2026-08-15T06:09:25.789897Z","submitted_at":"2025-03-06T18:43:29Z","title":"L1: Controlling How Long A Reasoning Model Thinks With Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.04697","snapshot_observed_at":"2026-08-06T17:35:30.988070Z","title":"L1: Controlling how long a reasoning model thinks with reinforcement learning.arXiv preprint arXiv:2503.04697, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:30.988070Z"},"links":{"cited_paper":"/paper/2503.04697","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:0ededd97e40654911dfcda4f7f043b8b5090bd88ae609fd4a4e77532be5fca3a","observation_id":"45263c56-4939-4fd5-a35f-802585d0b16f","resolution":{"observed_at":"2026-08-06T17:35:30.988070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:34.121167Z","title":"AIMO Validation AIME Dataset","venue":null,"work_id":"49269060-daef-49ca-b7e3-1d19ebf632c7","year":null},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.077245Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:08b5104a994c2f65d38caf9a84e6c7d5bb2be0fb9ec3a540098d07c3e8440432","observation_id":"1a4ed58e-988f-41b8-88fb-1d88bd327e52","resolution":{"observed_at":"2026-08-06T17:35:34.128623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:31.201122Z","title":"Training language models to reason efficiently.arXiv preprint arXiv:2502.04463, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.201122Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:c583d0b6c18d706120a38abc137fe076cc4335a3432002aec5766c7cbe2729fc","observation_id":"2746420f-7962-4a4a-8836-444cd63e6cc4","resolution":{"observed_at":"2026-08-06T17:35:31.201122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:31.325011Z","title":"Llama-nemotron: Efficient reasoning models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.325011Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:85e89d40787e0ff5650a20c3bbb87959cd413761ecbd3b2018e5c1ac9e71ca06","observation_id":"ebfce389-c508-42f1-9643-13c3c3f797ea","resolution":{"observed_at":"2026-08-06T17:35:31.325011Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.21187","last_updated":"2025-02-01T07:57:37Z","snapshot_observed_at":"2026-08-01T16:43:44.704797Z","submitted_at":"2024-12-30T18:55:12Z","title":"Do NOT Think That Much for 2+3=? On the Overthinking of o1-Like LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.21187","snapshot_observed_at":"2026-08-06T17:35:31.347313Z","title":"Do not think that much for 2+ 3=? on the overthinking of o1-like llms.arXiv preprint arXiv:2412.21187, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.347313Z"},"links":{"cited_paper":"/paper/2412.21187","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:f1ea980c0b4dd56156ad4b280dec955423999710949fff075ecb96daecf95a52","observation_id":"7180110f-4272-4077-be7d-091de1f211b1","resolution":{"observed_at":"2026-08-06T17:35:31.347313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:34.094159Z","title":"Batch prompting: Efficient inference with large language model apis","venue":null,"work_id":"691c0895-7419-48e2-88ec-508ed92937d6","year":2023},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.355798Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:9614385663cd60a46190b6418b4680fb350271af480bd48fbf7def944df76e94","observation_id":"c161d8cb-73d7-4eb3-ae87-d10ad418edba","resolution":{"observed_at":"2026-08-06T17:35:34.104687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-06T17:35:31.373974Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.373974Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:7b3ccc0989079e97f671edb185b9280715ad744df9a631bb88f7aff6833aed87","observation_id":"a058e47c-cbf6-4ccb-aea2-2ad48f7d49e4","resolution":{"observed_at":"2026-08-06T17:35:31.373974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01456","last_updated":"2025-09-26T09:25:31Z","snapshot_observed_at":"2026-08-17T04:59:59.434643Z","submitted_at":"2025-02-03T15:43:48Z","title":"Process Reinforcement through Implicit Rewards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01456","snapshot_observed_at":"2026-08-06T17:35:31.381615Z","title":"Process reinforcement through implicit rewards.arXiv preprint arXiv:2502.01456, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.381615Z"},"links":{"cited_paper":"/paper/2502.01456","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:9652285fa96377277797c3beda62e41864dc77a65600fa0305702e5b170d2689","observation_id":"5f70281e-45cc-43cf-8302-8ba8372d1a95","resolution":{"observed_at":"2026-08-06T17:35:31.381615Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:34.072055Z","title":"Open r1: A fully open reproduction of deepseek-r1, January 2025","venue":null,"work_id":"38bcf2ab-ad84-4f30-8bb0-e60424133634","year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.400902Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:ed460c335dd77ff555acd73014bee0e153a89d361276c3058c9456cd9819472a","observation_id":"d41e278d-97de-4f5e-9cb1-3cbaab6e17ac","resolution":{"observed_at":"2026-08-06T17:35:34.077786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01307","last_updated":"2025-08-15T15:21:46Z","snapshot_observed_at":"2026-08-14T13:46:28.086398Z","submitted_at":"2025-03-03T08:46:22Z","title":"Cognitive Behaviors that Enable Self-Improving Reasoners, or, Four Habits of Highly Effective STaRs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01307","snapshot_observed_at":"2026-08-06T17:35:31.408154Z","title":"Cognitive behaviors that enable self-improving reasoners, or, four habits of highly effective stars.arXiv preprint arXiv:2503.01307, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.408154Z"},"links":{"cited_paper":"/paper/2503.01307","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:dfcb0ab9f019adf65b6031f76d81de118428621a15f82f79d4524964a81c4ee2","observation_id":"de32417e-e383-4e0a-b2b9-456d3a002b82","resolution":{"observed_at":"2026-08-06T17:35:31.408154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07985","last_updated":"2024-12-24T04:04:30Z","snapshot_observed_at":"2026-08-16T19:47:24.627230Z","submitted_at":"2024-10-10T14:39:33Z","title":"Omni-MATH: A Universal Olympiad Level Mathematic Benchmark For Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07985","snapshot_observed_at":"2026-08-06T17:35:31.417001Z","title":"Omni-math: A universal olympiad level mathematic benchmark for large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.417001Z"},"links":{"cited_paper":"/paper/2410.07985","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:d497fd82283abf37cf590541e8a368be9fed9212ece4a1095adc198978ac8707","observation_id":"06093aef-6291-4c7e-aa5b-3a6027b33ad1","resolution":{"observed_at":"2026-08-06T17:35:31.417001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T17:35:31.423282Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.423282Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:72ebc974f29d4b395f042785985bfeea3f633ebc881d4bc52a7981ac10cabb01","observation_id":"bd95410d-82d2-4ac5-96a3-ae684348192f","resolution":{"observed_at":"2026-08-06T17:35:31.423282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.18547","last_updated":"2025-06-02T00:44:09Z","snapshot_observed_at":"2026-08-16T12:59:58.216334Z","submitted_at":"2024-12-24T16:55:45Z","title":"Token-Budget-Aware LLM Reasoning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.18547","snapshot_observed_at":"2026-08-06T17:35:31.430010Z","title":"Token-budget- aware llm reasoning.arXiv preprint arXiv:2412.18547, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.430010Z"},"links":{"cited_paper":"/paper/2412.18547","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:0aa336f4f9458d42c5cb82fc84595dd28f6e209ff4f77b7cf6423860837d503f","observation_id":"c2cf8936-1197-459c-ae5c-e25a64c8a1d5","resolution":{"observed_at":"2026-08-06T17:35:31.430010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:34.042886Z","title":"Olympiadbench: A challenging benchmark for promoting agi with olympiad- level bilingual multimodal scientific problems","venue":null,"work_id":"ccc29488-2836-4114-8a5a-4e376c5b31d5","year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.435294Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:54c1a6cade3dae8aaee0c1ee782aab68a9f790bb80cc04b59f38d1f2c3711f27","observation_id":"47fbd123-0066-482a-a30e-1246a88e7a83","resolution":{"observed_at":"2026-08-06T17:35:34.047658Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:34.005536Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":"ee3f76d4-a102-4599-aad5-1f95e092a067","year":2021},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.440805Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:0b5077a726b5e6e18bed1adfd7904dd879873a630fd5792d481a2de0bd582153","observation_id":"6b20a37a-2a46-42b7-bc35-26cacc39a33a","resolution":{"observed_at":"2026-08-06T17:35:34.018469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-06T17:35:31.446614Z","title":"Measuring mathematical problem solving with the math dataset, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.446614Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:e647548c338d112788d5793589992f41973bb907913294768c5ebb18a51f7492","observation_id":"ba63a264-19df-4168-a4f4-5a6159ecca68","resolution":{"observed_at":"2026-08-06T17:35:31.446614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:31.463634Z","title":"A sober look at progress in language model reasoning: Pitfalls and paths to reproducibility.arXiv preprint arXiv:2504.07086, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.463634Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:4fc1d00fa2b80919c0728243e846dadd227eb3fb41a21d88f7d11e99aea40582","observation_id":"10fd9226-c8f3-484c-b213-a18a7980bd64","resolution":{"observed_at":"2026-08-06T17:35:31.463634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:31.469333Z","title":"Compound-qa: A benchmark for evaluating llms on compound questions.arXiv preprint arXiv:2411.10163, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.469333Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:fafc8c2c9e4edca0320ad01c3f92f60c01819b23149ed347aec81b97599162ee","observation_id":"c167a1d1-345d-4cc1-9172-f02160a6a1e8","resolution":{"observed_at":"2026-08-06T17:35:31.469333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24290","last_updated":"2025-07-05T09:01:04Z","snapshot_observed_at":"2026-08-12T18:49:18.486177Z","submitted_at":"2025-03-31T16:36:05Z","title":"Open-Reasoner-Zero: An Open Source Approach to Scaling Up Reinforcement Learning on the Base Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24290","snapshot_observed_at":"2026-08-06T17:35:31.474664Z","title":"Open- reasoner-zero: An open source approach to scaling up reinforcement learning on the base model.arXiv preprint arXiv:2503.24290, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.474664Z"},"links":{"cited_paper":"/paper/2503.24290","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:3339b2c5c2d3329b513379e5a60b41e0515466995f6f83c183372526f18b4c5d","observation_id":"5374261a-fc43-48f6-8a2d-89c639e966ec","resolution":{"observed_at":"2026-08-06T17:35:31.474664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12186","last_updated":"2024-11-12T13:24:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:57:57Z","title":"Qwen2.5-Coder Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12186","snapshot_observed_at":"2026-08-06T17:35:31.479733Z","title":"Qwen2.5-coder technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.479733Z"},"links":{"cited_paper":"/paper/2409.12186","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:c968f4911b86f482dbebc124de71bab881a3e6ec1391b47d3e6e01385f1a3ba7","observation_id":"d9b75dce-a692-4a17-96c3-e4c1b56b8201","resolution":{"observed_at":"2026-08-06T17:35:31.479733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07974","last_updated":"2024-06-06T17:41:21Z","snapshot_observed_at":"2026-08-16T07:05:57.323612Z","submitted_at":"2024-03-12T17:58:04Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07974","snapshot_observed_at":"2026-08-06T17:35:31.486293Z","title":"Livecodebench: Holistic and contamination free evaluation of large language models for code.arXiv preprint arXiv:2403.07974, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.486293Z"},"links":{"cited_paper":"/paper/2403.07974","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:a94d97bd0a0b4d51269ac49cd3311e40ea270c10dc1a7e3a013425e1973557a5","observation_id":"8827ea6b-82b3-4512-8c17-6d9418c39241","resolution":{"observed_at":"2026-08-06T17:35:31.486293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.965400Z","title":"Swe-bench: Can language models resolve real-world github issues? InThe Twelfth International Conference on Learning Representations","venue":null,"work_id":"a29b7021-c38a-4300-a5dd-71024fe7ff59","year":null},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.493366Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:26f0c4686b3086b9f47658dc2ec942d26c2a346d4893f79b82e0fee71acaa4d0","observation_id":"f8d00e6a-8f2d-43fe-b230-6fbfe265d4b8","resolution":{"observed_at":"2026-08-06T17:35:33.973252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.13326","last_updated":"2025-06-05T01:47:25Z","snapshot_observed_at":"2026-08-16T17:50:13.035368Z","submitted_at":"2024-05-22T04:08:20Z","title":"Mosaic-IT: Cost-Free Compositional Data Synthesis for Instruction Tuning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.13326","snapshot_observed_at":"2026-08-06T17:35:31.498612Z","title":"Mosaic-it: Free compositional data augmentation improves instruction tuning.arXiv preprint arXiv:2405.13326, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.498612Z"},"links":{"cited_paper":"/paper/2405.13326","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:85582cdc559413db8a919681d197743610c0ea1eb9695a8c8c5ec8b1c9d1caec","observation_id":"fbf33416-9f5c-453b-89b2-20a4bfe2f7ac","resolution":{"observed_at":"2026-08-06T17:35:31.498612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.19093","last_updated":"2025-04-27T03:41:17Z","snapshot_observed_at":"2026-08-16T05:59:22.377813Z","submitted_at":"2025-04-27T03:41:17Z","title":"CipherBank: Exploring the Boundary of LLM Reasoning Capabilities through Cryptography Challenges","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.19093","snapshot_observed_at":"2026-08-06T17:35:31.504727Z","title":"Cipherbank: Exploring the boundary of llm reasoning capabilities through cryptography challenges.arXiv preprint arXiv:2504.19093, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.504727Z"},"links":{"cited_paper":"/paper/2504.19093","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:ec70f766e6556322632aaffb28aef0a1885cac5010a356e37ae1b3a5388ba8fc","observation_id":"def049e9-6de0-4ab7-8a45-1410b4ad7429","resolution":{"observed_at":"2026-08-06T17:35:31.504727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14891","last_updated":"2025-03-19T04:36:35Z","snapshot_observed_at":"2026-08-16T12:48:45.451621Z","submitted_at":"2025-03-19T04:36:35Z","title":"MetaLadder: Ascending Mathematical Solution Quality via Analogical-Problem Reasoning Transfer","version":1},"cited_work":{"arxiv_id":"2503.14891","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.14891","snapshot_observed_at":"2026-08-06T17:35:32.533458Z","title":"MetaLadder: Ascending Mathematical Solution Quality via Analogical-Problem Reasoning Transfer","venue":"cs.CL","work_id":"6fa775dc-1f67-458e-a358-50d00a5673ab","year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.510786Z"},"links":{"cited_paper":"/paper/2503.14891","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:e4272b51429991e3f84d35f043d637c94df3d1bf066eab8cb0859d0216db1c52","observation_id":"64f47877-0a6e-4bd5-878e-343153bdd0d2","resolution":{"observed_at":"2026-08-06T17:35:32.543004Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.937656Z","title":"Aime 2025 dataset, 2025","venue":null,"work_id":"9e3971f0-eee7-48ef-bd5d-ec03c315fc37","year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.517198Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:a71f19907b6bf3067e720dd8795b89ccee1378ad5d7aedffd43ef7adcec645d1","observation_id":"0c557471-d394-4ea3-b7cf-7aef5ebf7d00","resolution":{"observed_at":"2026-08-06T17:35:33.947508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.912821Z","title":"Lost in the middle: How language models use long contexts.T ransactions of the Association for Computational Linguistics, 12, 2024","venue":null,"work_id":"67b3813a-7089-444a-bc99-281fcb72e05e","year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.524111Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:c3dced908763cc7848f96f35b81dcbd6ea41ae5fdd3996151e9b468d76931aa1","observation_id":"b990de15-63eb-4dcd-a11e-cf6bbba9e77f","resolution":{"observed_at":"2026-08-06T17:35:33.918264Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12570","last_updated":"2025-01-29T03:11:03Z","snapshot_observed_at":"2026-08-10T17:00:34.555363Z","submitted_at":"2025-01-22T01:35:11Z","title":"O1-Pruner: Length-Harmonizing Fine-Tuning for O1-Like Reasoning Pruning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12570","snapshot_observed_at":"2026-08-06T17:35:31.531602Z","title":"O1-pruner: Length-harmonizing fine-tuning for o1-like reasoning pruning.arXiv preprint arXiv:2501.12570, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.531602Z"},"links":{"cited_paper":"/paper/2501.12570","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:4c51c4e3990e5bf9185e528bddfdc15d36a9e2762f01fd34f9bba8661fad2ca0","observation_id":"5204b94e-7408-418d-b2d9-cda9cc9501f7","resolution":{"observed_at":"2026-08-06T17:35:31.531602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.879533Z","title":"Tang, Manan Roongta, Colin Cai, Jeffrey Luo, Li Erran Li, Raluca Ada Popa, and Ion Stoica","venue":null,"work_id":"24a0746f-b04b-4a15-913e-97ffc53f7635","year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.537341Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:bca8b5328e22545c8bce5083fccf0933ca1825434a01f780da746dfd9f0d1a21","observation_id":"36a9691b-8521-436f-a46e-944e0bcf5b58","resolution":{"observed_at":"2026-08-06T17:35:33.887993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:31.544110Z","title":"Real: Efficient rlhf training of large language models with parameter reallocation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.544110Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:588ff237af0b53b0abc2093af082b7cef1e9586d924daaf5069f743920fa9257","observation_id":"f1118e6a-569d-4c15-ab13-16284aed28a1","resolution":{"observed_at":"2026-08-06T17:35:31.544110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.19393","last_updated":"2025-03-01T06:07:39Z","snapshot_observed_at":"2026-08-14T11:38:51.383664Z","submitted_at":"2025-01-31T18:48:08Z","title":"s1: Simple test-time scaling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.19393","snapshot_observed_at":"2026-08-06T17:35:31.550519Z","title":"s1: Simple test-time scaling, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.550519Z"},"links":{"cited_paper":"/paper/2501.19393","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:659076326069db66725919f6edd86f5cfeface982282ee384c07e1be7df7cf50","observation_id":"3e452e6f-618b-4eac-b43e-04b4ac0359f8","resolution":{"observed_at":"2026-08-06T17:35:31.550519Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.19825","last_updated":"2025-01-23T08:45:52Z","snapshot_observed_at":"2026-08-16T13:30:18.557142Z","submitted_at":"2024-07-29T09:21:52Z","title":"Concise Thoughts: Impact of Output Length on LLM Reasoning and Cost","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.19825","snapshot_observed_at":"2026-08-06T17:35:31.557650Z","title":"Concise thoughts: Impact of output length on llm reasoning and cost.arXiv preprint arXiv:2407.19825, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.557650Z"},"links":{"cited_paper":"/paper/2407.19825","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:3481492f1eb201fcb53289b9423a38d354a98963fc610e291a1b06846240b4b6","observation_id":"d41f9fc1-155e-455a-83e2-f47493bfacc5","resolution":{"observed_at":"2026-08-06T17:35:31.557650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.818148Z","title":"Openai o3 and o4-mini system card, Apr 2025","venue":null,"work_id":"2e040dc1-a0b9-44cd-93d2-5be1ab2f16bb","year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.573060Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:545870301498b67f51a7a425796a173a58c85ddd9f0ff265f8d8f1458cabb716","observation_id":"39780489-485f-4189-bbc1-fd9062f744c1","resolution":{"observed_at":"2026-08-06T17:35:33.841219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.17439","last_updated":"2025-05-30T15:19:51Z","snapshot_observed_at":"2026-08-16T12:47:51.733157Z","submitted_at":"2025-03-21T17:59:10Z","title":"LEMMA: Learning from Errors for MatheMatical Advancement in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.17439","snapshot_observed_at":"2026-08-06T17:35:31.578289Z","title":"Lemma: Learning from errors for mathematical advancement in llms.arXiv preprint arXiv:2503.17439, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.578289Z"},"links":{"cited_paper":"/paper/2503.17439","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:029cf1f9e9d6195a04eb24757fb090973bc86d31a5e67a9a53520cebdd6c70ff","observation_id":"0a7d6353-7a6c-4cce-9c82-e62afc5fc3ee","resolution":{"observed_at":"2026-08-06T17:35:31.578289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.16212","last_updated":"2025-06-16T05:58:00Z","snapshot_observed_at":"2026-08-16T12:48:15.363075Z","submitted_at":"2025-03-20T15:00:41Z","title":"MathFusion: Enhancing Mathematical Problem-solving of LLM through Instruction Fusion","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.16212","snapshot_observed_at":"2026-08-06T17:35:31.583773Z","title":"Mathfusion: Enhancing mathematic problem-solving of llm through instruction fusion.arXiv preprint arXiv:2503.16212, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.583773Z"},"links":{"cited_paper":"/paper/2503.16212","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:29ea94d251629b962f81208acb1d2c0eb5a5ef6c49a6b181a457b508508ce521","observation_id":"f3315be9-bdf9-4c7a-a7b7-0a90b1e52275","resolution":{"observed_at":"2026-08-06T17:35:31.583773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01257","last_updated":"2025-01-03T16:36:12Z","snapshot_observed_at":"2026-08-15T21:08:10.475006Z","submitted_at":"2025-01-02T13:49:00Z","title":"CodeElo: Benchmarking Competition-level Code Generation of LLMs with Human-comparable Elo Ratings","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01257","snapshot_observed_at":"2026-08-06T17:35:31.589921Z","title":"Codeelo: Benchmarking competition-level code generation of llms with human-comparable elo ratings.arXiv preprint arXiv:2501.01257, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.589921Z"},"links":{"cited_paper":"/paper/2501.01257","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:1eedd1eb507f302e7e18dea65630484c4917bcd9d00970bae6a7b33ee9287c05","observation_id":"bce54451-14c6-4b2c-aa51-696bf63c5a03","resolution":{"observed_at":"2026-08-06T17:35:31.589921Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:31.595871Z","title":"Gpqa: A graduate-level google-proof q&a benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.595871Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:56e0a16b8b794200bafd517c267586d9812f7bd5f28a691d2a7a8d74c7a9d743","observation_id":"ba7dca3b-95c3-4ad4-bb5f-c690046d5165","resolution":{"observed_at":"2026-08-06T17:35:31.595871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12950","last_updated":"2024-01-31T19:47:26Z","snapshot_observed_at":"2026-08-13T04:25:38.282910Z","submitted_at":"2023-08-24T17:39:13Z","title":"Code Llama: Open Foundation Models for Code","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12950","snapshot_observed_at":"2026-08-06T17:35:31.600900Z","title":"Code llama: Open foundation models for code, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.600900Z"},"links":{"cited_paper":"/paper/2308.12950","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:f4242e388c4bf09f26a4320c44b55da9b88160a0f92b0926657f55e3a2a3a84a","observation_id":"4cb57203-05da-4b9f-b0ed-a5a09edc8c99","resolution":{"observed_at":"2026-08-06T17:35:31.600900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.762651Z","title":"A practitioners’ guide to transfer learning for text classification using convolutional neural networks","venue":null,"work_id":"946e4b48-2e04-48c3-acba-a63b0cfb5359","year":2018},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.606428Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:3438651aa71d4fa5033e4e2f5dbcdac050b81688159440606e0bdb428dcad3de","observation_id":"7b37e65d-e30c-4cab-b8ed-574fb9b42aea","resolution":{"observed_at":"2026-08-06T17:35:33.777088Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04022","last_updated":"2025-04-05T02:24:07Z","snapshot_observed_at":"2026-08-16T12:43:48.435094Z","submitted_at":"2025-04-05T02:24:07Z","title":"Rethinking Reflection in Pre-Training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.04022","snapshot_observed_at":"2026-08-06T17:35:31.611579Z","title":"Rethinking reflection in pre-training.arXiv preprint arXiv:2504.04022, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.611579Z"},"links":{"cited_paper":"/paper/2504.04022","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:7af6661ac946cd4711a048681b9095cc00081e0c8985c3f9969c404cb8a0bcd5","observation_id":"1b4d177f-3de1-4cb5-a6d6-90521a19064a","resolution":{"observed_at":"2026-08-06T17:35:31.611579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-06T17:35:31.617129Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.617129Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:419d555e685d4653a8bbe7924f1455e1394594166943eb73e468490a2bb046e9","observation_id":"6f9f2f83-dde7-48ee-8ea0-60bc1d38bb34","resolution":{"observed_at":"2026-08-06T17:35:31.617129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.11061","last_updated":"2024-08-07T19:32:59Z","snapshot_observed_at":"2026-08-16T13:27:52.057441Z","submitted_at":"2024-08-07T19:32:59Z","title":"StructuredRAG: JSON Response Formatting with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.11061","snapshot_observed_at":"2026-08-06T17:35:31.623431Z","title":"Structuredrag: Json response formatting with large language models.arXiv preprint arXiv:2408.11061, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.623431Z"},"links":{"cited_paper":"/paper/2408.11061","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:62d2720865706c5870a51f102a6a02985c4fedb05683c919f906acde4a84d26a","observation_id":"2681144d-be27-4ec2-9346-bf4175310e8e","resolution":{"observed_at":"2026-08-06T17:35:31.623431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.726580Z","title":null,"venue":null,"work_id":"f1632034-056f-4e03-9213-aec83cf98a4e","year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.629984Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:764ecd47422027f9b618dbcd7dac6ac5a74b4209204d39ed7bcab148f11e2079","observation_id":"1f1595e1-a2cb-4814-bf27-bb3bea871c58","resolution":{"observed_at":"2026-08-06T17:35:33.734942Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.16419","last_updated":"2025-08-21T19:14:40Z","snapshot_observed_at":"2026-08-11T13:10:23.709172Z","submitted_at":"2025-03-20T17:59:38Z","title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.16419","snapshot_observed_at":"2026-08-06T17:35:31.635546Z","title":"Stop overthinking: A survey on efficient reasoning for large language models.arXiv preprint arXiv:2503.16419, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.635546Z"},"links":{"cited_paper":"/paper/2503.16419","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:f314baee41efe97f981ac6086406a5f78991b50c7efd0addba50f2ea6fdb4332","observation_id":"44f19072-1194-48ea-8190-4c249b62aa5a","resolution":{"observed_at":"2026-08-06T17:35:31.635546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.693350Z","title":"Commonsenseqa: A question answer- ing challenge targeting commonsense knowledge","venue":null,"work_id":"aab3e7c5-d553-4db4-9923-f2357ad4bd38","year":2019},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.641299Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:0305f4698b43b09964dcd50e17a09667bfe5944f888d77d66e7398885cb33cab","observation_id":"bca4952e-b9f3-4f99-b3da-19b652ad3bb7","resolution":{"observed_at":"2026-08-06T17:35:33.702329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.02442","last_updated":"2024-10-14T13:57:29Z","snapshot_observed_at":"2026-08-16T13:28:29.726326Z","submitted_at":"2024-08-05T13:08:24Z","title":"Let Me Speak Freely? A Study on the Impact of Format Restrictions on Performance of Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.02442","snapshot_observed_at":"2026-08-06T17:35:31.645965Z","title":"Let me speak freely? a study on the impact of format restrictions on performance of large language models.arXiv preprint arXiv:2408.02442, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.645965Z"},"links":{"cited_paper":"/paper/2408.02442","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:17abe57aa559062ee907ee033cbf40eba8a8838234d2c1923130b293d12bd3f9","observation_id":"3592653e-52be-4764-ac17-dbde88ab955f","resolution":{"observed_at":"2026-08-06T17:35:31.645965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-06T17:35:31.651233Z","title":"Gemini: a family of highly capable multimodal models.arXiv preprint arXiv:2312.11805, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.651233Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:19e9d5b03e2c300271e63d3037998e25927bf1414a3ed3705a45d3fa367b311b","observation_id":"4c333033-5f40-49a3-8bdb-7d6c65dfd001","resolution":{"observed_at":"2026-08-06T17:35:31.651233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19786","last_updated":"2025-03-25T15:52:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-25T15:52:34Z","title":"Gemma 3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19786","snapshot_observed_at":"2026-08-06T17:35:31.656844Z","title":"Gemma 3 technical report.arXiv preprint arXiv:2503.19786, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.656844Z"},"links":{"cited_paper":"/paper/2503.19786","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:fc388ad3945690a8fc5301bfc8b996ca2782dcfd4650eb0cff0ba802a8adb1a2","observation_id":"04ed657f-18a9-4afa-884e-b8c9effaf7d8","resolution":{"observed_at":"2026-08-06T17:35:31.656844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12599","last_updated":"2025-06-03T02:14:54Z","snapshot_observed_at":"2026-08-15T22:38:53.825110Z","submitted_at":"2025-01-22T02:48:14Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12599","snapshot_observed_at":"2026-08-06T17:35:31.662296Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.662296Z"},"links":{"cited_paper":"/paper/2501.12599","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:18752f6018aa6c3a91a7b0f163996b7fd56dc4487d0911c0a09e6714ddcde60b","observation_id":"f127b587-d06f-4894-8541-986dae1a3acb","resolution":{"observed_at":"2026-08-06T17:35:31.662296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.664795Z","title":"Open Thoughts, January 2025","venue":null,"work_id":"c4162684-8af0-4300-a453-67d2d61d0a69","year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.668855Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:8361a9c314d07d124a3e09cfd02a7841905af0b4429b10d7be9f7ee64bc29f99","observation_id":"9f0747c9-feec-49a6-a2ff-1d24a093019d","resolution":{"observed_at":"2026-08-06T17:35:33.670553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:31.674386Z","title":"Qwq-32b: Embracing the power of reinforcement learning, March 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.674386Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:2792d8f8efcea81cdc7e683c85422efdf8e583ae364065239495ae502356a680","observation_id":"7e17e468-f86e-45dc-8610-5634a5272ccb","resolution":{"observed_at":"2026-08-06T17:35:31.674386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18585","last_updated":"2025-02-18T16:51:53Z","snapshot_observed_at":"2026-08-15T06:19:06.009960Z","submitted_at":"2025-01-30T18:58:18Z","title":"Thoughts Are All Over the Place: On the Underthinking of o1-Like LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18585","snapshot_observed_at":"2026-08-06T17:35:31.681214Z","title":"Thoughts are all over the place: On the underthinking of o1-like llms.arXiv preprint arXiv:2501.18585, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.681214Z"},"links":{"cited_paper":"/paper/2501.18585","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:631df2de8e0f0e57b8d9bf2511aa9f69fbaa89a8f7c1f7bd711151f82bdc8ac0","observation_id":"df5bbf48-696d-48e1-ad0a-e5f70cfe8bd2","resolution":{"observed_at":"2026-08-06T17:35:31.681214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.620162Z","title":"Evaluating llms with multiple problems at once: A new paradigm for probing llm capabilities.arXiv e-prints, pages arXiv–2406, 2024","venue":null,"work_id":"67665b72-95ca-43b7-a398-e20ba2ecab3f","year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.686947Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:0e0a7689cac397212131b5286b51946f6acbae9f9f7f79766a35b7178a3dd921","observation_id":"735fa473-68b5-4aae-b2d0-b8187d8135bd","resolution":{"observed_at":"2026-08-06T17:35:33.631351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-16T12:50:20.333530Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-06T17:35:31.692436Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.692436Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:21495029930969e3b051fe14ea983952951f681887ae45236206175e33f89827","observation_id":"b60c03f1-14c4-4a95-a520-91ce8f4fccf3","resolution":{"observed_at":"2026-08-06T17:35:31.692436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-06T17:35:31.698122Z","title":"Qwen2.5 technical report.arXiv preprint arXiv:2412.15115, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.698122Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:e3c56ff103cbf7d6b8dab97b328a6099ce03fb37a51708aeb14817c0db875604","observation_id":"5775c902-e633-4850-8d68-933baa18105b","resolution":{"observed_at":"2026-08-06T17:35:31.698122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12122","last_updated":"2024-09-18T16:45:37Z","snapshot_observed_at":"2026-08-14T16:00:41.902820Z","submitted_at":"2024-09-18T16:45:37Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12122","snapshot_observed_at":"2026-08-06T17:35:31.712672Z","title":"Qwen2.5-math technical report: Toward mathematical expert model via self-improvement.arXiv preprint arXiv:2409.12122, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.712672Z"},"links":{"cited_paper":"/paper/2409.12122","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:09149ed95e3d6a1bfbbb2bdfcebb5293d6e7c553514e4174bbe5affc8fb8ee49","observation_id":"ef69f0ad-aad0-4073-b79f-7f9ea4ac61cf","resolution":{"observed_at":"2026-08-06T17:35:31.712672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.596674Z","title":"Aime-preview: A rigorous and immediate evalua- tion framework for advanced mathematical reasoning","venue":null,"work_id":"0d1233da-f8e4-4dff-926e-e2b43a59f9ec","year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.718497Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:9ba7dba8a690866529dfcd59aaf48432ebb650d3b4b2329db332eabfbfe71d94","observation_id":"111b102f-a05e-4ae1-bf88-8700d95d4e91","resolution":{"observed_at":"2026-08-06T17:35:33.605064Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.572900Z","title":"Mitigate position bias in large language models via scaling a single dimension","venue":null,"work_id":"50a416ed-7a2e-4295-86cd-bde44999df5b","year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.724414Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:66bd0675ade11c13070ef20829e2598e824d84f3be6a336e4b8be7b7f5452473","observation_id":"a62d3ed8-d201-482f-a455-a323b0323cab","resolution":{"observed_at":"2026-08-06T17:35:33.581735Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13837","last_updated":"2025-11-24T06:11:04Z","snapshot_observed_at":"2026-08-14T14:03:15.178702Z","submitted_at":"2025-04-18T17:59:56Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.13837","snapshot_observed_at":"2026-08-06T17:35:31.731325Z","title":"Does reinforcement learning really incentivize reasoning capacity in llms beyond the base model?arXiv preprint arXiv:2504.13837, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.731325Z"},"links":{"cited_paper":"/paper/2504.13837","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:3a41c2a907c8c6e6080416f21ed3a4225d8646fba5f3f8670b36b7b31feff6b7","observation_id":"1b4084d5-1f64-4f1e-97a8-33d14e2df46d","resolution":{"observed_at":"2026-08-06T17:35:31.731325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18892","last_updated":"2025-08-06T08:42:32Z","snapshot_observed_at":"2026-07-06T20:57:57.039376Z","submitted_at":"2025-03-24T17:06:10Z","title":"SimpleRL-Zoo: Investigating and Taming Zero Reinforcement Learning for Open Base Models in the Wild","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.18892","snapshot_observed_at":"2026-08-06T17:35:31.738286Z","title":"Simplerl-zoo: Investigating and taming zero reinforcement learning for open base models in the wild, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.738286Z"},"links":{"cited_paper":"/paper/2503.18892","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:1bdb6b4d5ded7d67b910bff1c9b5a4c5fb17983ea071eb61cc3d94a5689a8ee3","observation_id":"79d7c9bc-b709-45ee-80c2-f5b89985d4fb","resolution":{"observed_at":"2026-08-06T17:35:31.738286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14405","last_updated":"2024-11-25T17:57:55Z","snapshot_observed_at":"2026-08-12T15:10:50.113705Z","submitted_at":"2024-11-21T18:37:33Z","title":"Marco-o1: Towards Open Reasoning Models for Open-Ended Solutions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14405","snapshot_observed_at":"2026-08-06T17:35:31.748303Z","title":"Marco-o1: Towards open reasoning models for open-ended solutions, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.748303Z"},"links":{"cited_paper":"/paper/2411.14405","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:de13b8c6edce122519f2b061ed7da581adcfeee9eceeb6aee9e90eb535949ad6","observation_id":"20b70d40-304a-41d7-be02-63be69b72be2","resolution":{"observed_at":"2026-08-06T17:35:31.748303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.548395Z","title":"Your task is to extract the final answer from the prediction as it is, even if it is incorrect","venue":null,"work_id":"515b4ef0-c897-4ca5-81f7-dd05edd64050","year":null},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.759917Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:8f09aab8f40fc29b942251248939c1849144b3096fe954772a7a424c4ad8445a","observation_id":"77c51be8-50fa-40f3-bd10-77489d4565de","resolution":{"observed_at":"2026-08-06T17:35:33.556386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.516258Z","title":null,"venue":null,"work_id":"e53826cd-1fc9-4791-9fda-a796e25f8fc2","year":null},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.767438Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:b5b83b465ce01f5dab8f64da364ebe35da69ee3b283e10d8dae7326902d36531","observation_id":"331cc45b-32e9-4aa0-8c52-0f616cd45472","resolution":{"observed_at":"2026-08-06T17:35:33.530155Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.490265Z","title":"You should set the final answer to None (e.g., \\boxed{None})","venue":null,"work_id":"5fe5bc34-4a99-4d26-a79d-1165f1ff6b9f","year":null},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.779302Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:9272ae39b0289ec107eccda70d414082aa6f807f72f63b77f955a1884c239b0b","observation_id":"4c5a31fd-82c9-4ed0-b29b-3b8b337a5bbe","resolution":{"observed_at":"2026-08-06T17:35:33.496129Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.463881Z","title":null,"venue":null,"work_id":"50f5eb13-d475-4eff-bd29-f5c5630d4576","year":null},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.785220Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:13128ab31c93c5f1daea07a9b4498f68c6d90bf5127529289a5541aae607bdf2","observation_id":"ffbd914f-2835-49ec-aa56-1a60046abf66","resolution":{"observed_at":"2026-08-06T17:35:33.471313Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.435439Z","title":"For example, if there are three questions, the output should be Answer to Q1: \\boxed{answer 1} Answer to Q2: \\boxed{answer 2} Answer to Q3: \\boxed{answer 3}","venue":null,"work_id":"4cf3b13f-1b49-4ef1-be18-94ef5b54318f","year":null},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.790814Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:2241651e05e54bf4f94bd215dbd4d375db3ee876bbc059629f19d9ff206d2f7c","observation_id":"e76c610e-89fa-43b8-bbc6-310aca6f4d4f","resolution":{"observed_at":"2026-08-06T17:35:33.441592Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.393349Z","title":"We need to find this distance, express it in a specific form, and then compute m+n+p where the distance is m√n/p","venue":null,"work_id":"4b12b6a8-5f30-485e-ab48-7e8558361510","year":null},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.802279Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:18c375929153163d0f24892883da5ac35f53430ddd3d6a2b6b62131751df86b4","observation_id":"eb0aa1e9-157e-4a91-bf2b-81c7ca3f7619","resolution":{"observed_at":"2026-08-06T17:35:33.400304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.414136Z","title":"This distance can be written in the form m√n p , where m, n, and p are positive integers, m and p are relatively prime, andnis not divisible by the square of any prime","venue":null,"work_id":"99620537-4935-43ae-ad47-f63ef7a30303","year":null},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.796838Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:34b479e7e059917b0265e669f80e80ea8816031cc85f7b05ea752bc506ac6ed1","observation_id":"b69446d0-8b2e-4d0e-855c-f6703ce1eb05","resolution":{"observed_at":"2026-08-06T17:35:33.421756Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:35:33.356941Z","title":"tikz\\\"); label(\\","venue":null,"work_id":"6b530075-4dcc-4da2-a2bd-657b05c22917","year":2024},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":307,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.807793Z"},"links":{"citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:ea2e4d25bb69a49375a162424702f223263ddfef18fecf91fbe537152d74f095","observation_id":"9b5446c9-9f63-4f74-a0bd-2388734482a5","resolution":{"observed_at":"2026-08-06T17:35:33.371631Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-13T17:27:21.063855Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once"},"reference_resolution":{"displayed":69,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":46,"verified_exact":1,"verified_fuzzy":21},"total_outbound_references":69},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 69 of 69 outbound references and 5 inbound Pith citation observations for arXiv:2507.10541."}