{"as_of":"2026-08-10T07:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e72b64a42a3a41330350be58a23ab5bc3c8df199f22cff8281709f93d80037ed","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:29:03.805548Z","state":"measured"},{"denominator":64,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":64,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-03T21:20:00.041277Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T21:28:58.386085Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"cited_work":{"arxiv_id":"2507.02778","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02778","snapshot_observed_at":"2026-08-04T02:29:36.375484Z","title":"Self-correction bench: Uncovering and addressing the self-correction blind spot in large language models","venue":null,"work_id":"cdc79fb8-0919-4089-8a57-ab9adbd958d7","year":2025},"citing_paper":{"arxiv_id":"2605.05737","last_updated":"2026-05-07T06:29:34Z","snapshot_observed_at":"2026-07-06T23:18:17.333660Z","submitted_at":"2026-05-07T06:29:34Z","title":"ReFlect: An Effective Harness System for Complex Long-Horizon LLM Reasoning","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-08T11:35:25.204350Z"},"links":{"cited_paper":"/paper/2507.02778","citing_paper":"/paper/2605.05737"},"observation_digest":"sha256:37ede276f0d2159a53bb7ab5e4fd40a738c62ef4fd98b891815cd5d67c466ec7","observation_id":"3547e909-d01a-42bc-8fbe-10036c058fef","resolution":{"observed_at":"2026-08-04T02:29:36.375484Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"cited_work":{"arxiv_id":"2507.02778","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02778","snapshot_observed_at":"2026-08-04T02:29:36.375484Z","title":"Self-correction bench: Uncovering and addressing the self-correction blind spot in large language models","venue":null,"work_id":"cdc79fb8-0919-4089-8a57-ab9adbd958d7","year":2025},"citing_paper":{"arxiv_id":"2605.20867","last_updated":"2026-05-20T08:02:15Z","snapshot_observed_at":"2026-07-06T23:31:23.407780Z","submitted_at":"2026-05-20T08:02:15Z","title":"ProCrit: Self-Elicited Multi-Perspective Reasoning with Critic-Guided Revision for Multimodal Sarcasm Detection","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-21T02:35:37.724212Z"},"links":{"cited_paper":"/paper/2507.02778","citing_paper":"/paper/2605.20867"},"observation_digest":"sha256:e921769da61d7f81514b2a560579aaa001ccead53744f66daac72f4be7067f57","observation_id":"96d7ffc6-6dae-45a3-989d-37e0f37d0836","resolution":{"observed_at":"2026-08-04T02:29:36.375484Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"cited_work":{"arxiv_id":"2507.02778","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02778","snapshot_observed_at":"2026-08-04T02:29:36.375484Z","title":"Self-correction bench: Uncovering and addressing the self-correction blind spot in large language models","venue":null,"work_id":"cdc79fb8-0919-4089-8a57-ab9adbd958d7","year":2025},"citing_paper":{"arxiv_id":"2606.29425","last_updated":"2026-06-28T14:40:01Z","snapshot_observed_at":"2026-08-05T16:29:32.428034Z","submitted_at":"2026-06-28T14:40:01Z","title":"Mixture of Debaters: Learn to Debate at Architectural Level in Multi-Agent Reasoning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-30T07:11:02.464556Z"},"links":{"cited_paper":"/paper/2507.02778","citing_paper":"/paper/2606.29425"},"observation_digest":"sha256:70d5db48051538a5dcb25c60c9d3ef245faa71d8827e5f61f4d9d02963ae56aa","observation_id":"3a91d3a0-6f38-4726-acc8-06b923d8d159","resolution":{"observed_at":"2026-08-04T02:29:36.375484Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"cited_work":{"arxiv_id":"2507.02778","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02778","snapshot_observed_at":"2026-08-04T02:29:36.375484Z","title":"Self-correction bench: Uncovering and addressing the self-correction blind spot in large language models","venue":null,"work_id":"cdc79fb8-0919-4089-8a57-ab9adbd958d7","year":2025},"citing_paper":{"arxiv_id":"2607.02089","last_updated":"2026-07-01T14:25:43Z","snapshot_observed_at":"2026-08-02T16:57:02.105465Z","submitted_at":"2026-07-01T14:25:43Z","title":"ESC: Emotional Self-Correction for Reliable Vision-Language Models","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-07-03T21:20:00.041277Z"},"links":{"cited_paper":"/paper/2507.02778","citing_paper":"/paper/2607.02089"},"observation_digest":"sha256:6828c38cf358300415ce46280d1a9e4e6efa044a86792bf47d870e526baf8bbd","observation_id":"eaa9921f-da88-4f3c-9369-e2f5a7f177a6","resolution":{"observed_at":"2026-08-04T02:29:36.375484Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.02778/citation-record","integrity":"/paper/2507.02778/integrity","json":"/paper/2507.02778/citation-record.json","paper":"/paper/2507.02778"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T20:28:55.823384Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:55.823384Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:2a2fd7b1f4c827726ee185e1c0ccbbe97ed3bfc281d60ae8aa0e8179d844ec00","observation_id":"19a388be-1abf-4a33-ba82-7b35e40fd82f","resolution":{"observed_at":"2026-08-06T20:28:55.823384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:08.675442Z","title":"The claude 3 model family: Opus, sonnet, haiku, Mar 2024","venue":null,"work_id":"a1f4a303-4080-4221-8ca8-887d35541a91","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:55.944606Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:0f363a453790bf696777185eeea40356a0ec274e6405a116d3e4e9a91b580498","observation_id":"73f0cd9a-0b1b-408e-8072-9d94b04998e2","resolution":{"observed_at":"2026-08-06T20:29:08.781161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:08.375109Z","title":"Gemini 2.5: Pushing the frontier with advanced reasoning, multimodality, long context, and next generation agentic capabilities., June 2025","venue":null,"work_id":"74706def-875a-4674-8807-e696c5867219","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.097504Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:8e0b319dbe58a7f7c0affb31375cb1fb4492629ad7a2740a622df0ed1004a59f","observation_id":"50bbfc04-ce8c-4b4b-b5b0-3fff3e8b3f5a","resolution":{"observed_at":"2026-08-06T20:29:08.531479Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-06T20:28:56.227672Z","title":"Qwen3 technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.227672Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:3c205bd3938e661423213cfd44bd3ca0c2505bf3abd84eb36ded8d30b5d3c515","observation_id":"ae1989c8-73a6-49a0-8fc4-2d16699ff84d","resolution":{"observed_at":"2026-08-06T20:28:56.227672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:08.150401Z","title":"The llama 4 herd: The beginning of a new era of natively multimodal ai innovation, Apr 2025","venue":null,"work_id":"5b319c6b-8bd2-4a91-80bd-babcacb6c457","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.333821Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:19a9ee85c3796e709dd86b7b8e78deca984100cc29ea2f48dc23121714551733","observation_id":"586457e5-ab3b-4b4a-9464-8e20506298a9","resolution":{"observed_at":"2026-08-06T20:29:08.248924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T20:28:56.531856Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.531856Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:f885176b24b13b14a837444755b89ab6b04bbafc933fdc1573a6cd09c373a259","observation_id":"91ab01f4-557c-433b-b672-94dbef010bc0","resolution":{"observed_at":"2026-08-06T20:28:56.531856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:56.646944Z","title":"On faithfulness and factuality in abstractive summarization","venue":null,"work_id":null,"year":1906},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.646944Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:599e8449ba7440c3c00eee5856d0b58eb7f222a2439aad3bee20e65890ec9474","observation_id":"f38e80ac-bd0f-42de-b022-b8b7c6500e47","resolution":{"observed_at":"2026-08-06T20:28:56.646944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:56.783467Z","title":"A survey on hallucination in large language models: Principles, taxonomy, challenges, and open questions","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.783467Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:17f8e01e6ef59c7a8baa47b6efbea6a13abba67415b2c946038f281d567b743a","observation_id":"77338fbe-e90f-4bc5-bba8-76b029fc8ea4","resolution":{"observed_at":"2026-08-06T20:28:56.783467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:56.946027Z","title":"Do, Yan Xu, and Pascale Fung","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.946027Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:6df1d3709ad65a7448d988ad1abb11240e0e0ade829bb0781ecc93ded985d545","observation_id":"098af2f9-edf9-49c1-a735-ebf863240b3d","resolution":{"observed_at":"2026-08-06T20:28:56.946027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:57.071428Z","title":"Large language models can be easily distracted by irrelevant context","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.071428Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:9cc69970b6784bd1db549860b90f14f3a96f5943f3391ba516aa5e131335cd39","observation_id":"caa6fbd6-6cbc-4f13-b88a-1bcb810818e9","resolution":{"observed_at":"2026-08-06T20:28:57.071428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02061","last_updated":"2025-03-05T01:58:08Z","snapshot_observed_at":"2026-08-07T22:23:28.372208Z","submitted_at":"2024-06-04T07:43:33Z","title":"Alice in Wonderland: Simple Tasks Showing Complete Reasoning Breakdown in State-Of-the-Art Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02061","snapshot_observed_at":"2026-08-06T20:28:57.217863Z","title":"Alice in wonderland: Simple tasks showing complete reasoning breakdown in state-of-the-art large language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.217863Z"},"links":{"cited_paper":"/paper/2406.02061","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:9280b5c51ff1c49eb780e792d6c2b0ee3e80ca096fb65dbe147d76dbff4155c6","observation_id":"d87bcd25-bf69-48cd-b7fa-c14ce7548102","resolution":{"observed_at":"2026-08-06T20:28:57.217863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:07.881955Z","title":"Reflexion: language agents with verbal reinforcement learning","venue":null,"work_id":"ec724c1c-233d-4a90-b6ef-e04a5ac88a3a","year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.332058Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:4eb877f4c07d047812f34e2f29b079fa6a4eb9f78425591b932c12bd09268ee0","observation_id":"fae702c4-d4da-4075-8c5e-a0238f612dfe","resolution":{"observed_at":"2026-08-06T20:29:08.005937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:57.429687Z","title":"Self-refine: Iterative refinement with self-feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.429687Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:d1717ac1f906e9563b836221be1dadfeb2d0720a5b7314c83e76760020f668d2","observation_id":"1cecc2d2-7db8-48a5-94cf-cff9ceed325b","resolution":{"observed_at":"2026-08-06T20:28:57.429687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:07.590435Z","title":"Language models can solve computer tasks","venue":null,"work_id":"a220d232-22b8-45b3-91bd-ee74e3bc794f","year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.560434Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:5ec0376003bc572b06f66f77ffd3b062d33f10784f919791aff7931470ed9796","observation_id":"0952229d-8e2e-4192-9c73-59e1a4b6dcff","resolution":{"observed_at":"2026-08-06T20:29:07.724534Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:57.717223Z","title":"When can LLM s actually correct their own mistakes? a critical survey of self-correction of LLM s","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.717223Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:ca289b559eb817acd9e16def79b72fb7ba520c806e5c83510de1329bfd05ab2f","observation_id":"ee98ddf2-8acf-4313-9d25-4c6955fe54ee","resolution":{"observed_at":"2026-08-06T20:28:57.717223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01798","last_updated":"2024-03-14T04:27:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-03T04:56:12Z","title":"Large Language Models Cannot Self-Correct Reasoning Yet","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01798","snapshot_observed_at":"2026-08-06T20:28:57.838761Z","title":"Large language models cannot self-correct reasoning yet, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.838761Z"},"links":{"cited_paper":"/paper/2310.01798","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:06e862f32c272cd9ff4ce670631d89e538c5872e9a3e8b4b19c0ac07dfb69f09","observation_id":"7cb640df-e27c-4882-80c9-919eeba976dc","resolution":{"observed_at":"2026-08-06T20:28:57.838761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:57.998432Z","title":"LLM s cannot find reasoning errors, but can correct them given the error location","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.998432Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:6343e0ef45dfde26675d6da2800b513f25ae05a8ce9d7c1efdfae60e8943165e","observation_id":"8923266a-b790-4707-a354-bce566c6368d","resolution":{"observed_at":"2026-08-06T20:28:57.998432Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:07.333882Z","title":"Evaluating LLM s at detecting errors in LLM responses","venue":null,"work_id":"2cb19ed8-ba5b-43ad-81e8-940b0d113aa5","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.142786Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:f38788654b98943102cda5dc9ff69bdf927ed06f8b61ce36fe74b4f278022979","observation_id":"ed5bf506-a601-41b5-a800-544053498431","resolution":{"observed_at":"2026-08-06T20:29:07.454677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:07.103360Z","title":"Training language models to self-correct via reinforcement learning","venue":null,"work_id":"00012590-cccf-4ffb-901d-43484ac3ffa2","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.292638Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:e80ed1f090942303bf480822e547b300b659e5133adee775cc6d029523e8621f","observation_id":"5f1f55b3-3870-468a-9f67-1af9ce35a562","resolution":{"observed_at":"2026-08-06T20:29:07.214068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:06.774596Z","title":"Jailbroken: how does llm safety training fail? In Proceedings of the 37th International Conference on Neural Information Processing Systems, NIPS '23, Red Hook, NY, USA, 2023","venue":null,"work_id":"21c4619b-07e3-4ff7-b6be-600055be1f44","year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.454749Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:bbba26dc89bd60f5f9b90a283afe3f01305da5c1989c7aaa691650c3eba10fce","observation_id":"efddf2d7-1f69-4927-bb0a-f4f3896e609c","resolution":{"observed_at":"2026-08-06T20:29:06.943034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:06.452420Z","title":"Formalizing and benchmarking prompt injection attacks and defenses","venue":null,"work_id":"34c5eb3d-fefe-4d14-b227-e27febc8ec52","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.558097Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:6660afc97c0b0c1db9790d90c5fe275e498c7550a9cd49296fcd3972e55f68e7","observation_id":"553309ad-0b0f-4417-b8b1-f864b3625a67","resolution":{"observed_at":"2026-08-06T20:29:06.592654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13702","last_updated":"2023-07-17T01:08:39Z","snapshot_observed_at":"2026-07-31T04:21:27.501709Z","submitted_at":"2023-07-17T01:08:39Z","title":"Measuring Faithfulness in Chain-of-Thought Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13702","snapshot_observed_at":"2026-08-06T20:28:58.704896Z","title":"Bowman, and Ethan Perez","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.704896Z"},"links":{"cited_paper":"/paper/2307.13702","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:8a8bdfe8339fc808cd62cdbf2de3b607f6166ad8cb13b91abcb1a955fa2337bc","observation_id":"7caeb070-8c3f-4b21-a32e-ab3575a5ea7d","resolution":{"observed_at":"2026-08-06T20:28:58.704896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:06.139227Z","title":null,"venue":null,"work_id":"0144abe0-92f2-4653-9058-40959eaac168","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.869328Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:c34ba0c9f30210871638583aeb62d82b03e8c9f717474d3d4020867498badfae","observation_id":"9aced9d1-1d08-48f8-87f4-1e5b8e6a1a5b","resolution":{"observed_at":"2026-08-06T20:29:06.302848Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:59.074923Z","title":"Scaling LLM test-time compute optimally can be more effective than scaling parameters for reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.074923Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:a90fb2919450b9eaaebd4383ae7634db0fabf7421f64175ccbaa821a7cc50807","observation_id":"6e96adf5-08a3-472a-8d93-9315cfd851d5","resolution":{"observed_at":"2026-08-06T20:28:59.074923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.19393","last_updated":"2025-03-01T06:07:39Z","snapshot_observed_at":"2026-07-06T20:29:11.710285Z","submitted_at":"2025-01-31T18:48:08Z","title":"s1: Simple test-time scaling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.19393","snapshot_observed_at":"2026-08-06T20:28:59.270560Z","title":"s1: Simple test-time scaling, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.270560Z"},"links":{"cited_paper":"/paper/2501.19393","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:9d78565fec222e2993d3a6cb3aaa80893011f2972952a04a3d6708d3b72ee122","observation_id":"a61249e0-c3ef-455b-b871-6c8f7c79c70c","resolution":{"observed_at":"2026-08-06T20:28:59.270560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:59.400211Z","title":"Benchmarking cognitive biases in large language models as evaluators","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.400211Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:d6d26c9ba44d4a4b0ecaf2a5114a7e787922788dd5a1109be9e997b34a557453","observation_id":"f0954cbd-2c14-4b6b-8801-899f2a583ed1","resolution":{"observed_at":"2026-08-06T20:28:59.400211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:59.570749Z","title":"Cognitive bias in decision-making with LLM s","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.570749Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:9ff2e02c011bf2c2359c676194924a0a0a8134dea7deefc4dc8de3c03a14ebe3","observation_id":"78c11968-cc45-47a7-b086-2d93729aa853","resolution":{"observed_at":"2026-08-06T20:28:59.570749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:05.897396Z","title":"Capturing failures of large language models via human cognitive biases","venue":null,"work_id":"ef366fa0-dba3-4fd4-b94d-f026b5b698f1","year":2022},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.745163Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:b444eadecd9a1ded2ecef93f5f9954c790af3706fffadb795ed2afbeccce63c5","observation_id":"bc521204-141d-47ff-be90-5f34ec347f6e","resolution":{"observed_at":"2026-08-06T20:29:06.008527Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:59.906202Z","title":"Lin, and Lee Ross","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.906202Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:13deacf6e5a48cd305180fa3abd5afa1de4d677d8208eac7ed656bc02015ecec","observation_id":"42e7bac0-e04d-4130-8fc7-375e883a226c","resolution":{"observed_at":"2026-08-06T20:28:59.906202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06559","last_updated":"2025-05-26T14:03:32Z","snapshot_observed_at":"2026-08-09T05:18:18.356511Z","submitted_at":"2024-12-09T15:11:40Z","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06559","snapshot_observed_at":"2026-08-06T20:29:00.072960Z","title":"Processbench: Identifying process errors in mathematical reasoning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.072960Z"},"links":{"cited_paper":"/paper/2412.06559","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:44c11611282d452aaa46d6e9b2b0cc6d179d2b5ce855f618737dfb94dce27953","observation_id":"2c893750-911c-4989-a3ab-7264afd41c4f","resolution":{"observed_at":"2026-08-06T20:29:00.072960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.03124","last_updated":"2025-06-28T12:31:45Z","snapshot_observed_at":"2026-08-07T08:05:38.374743Z","submitted_at":"2025-01-06T16:31:45Z","title":"PRMBench: A Fine-grained and Challenging Benchmark for Process-Level Reward Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.03124","snapshot_observed_at":"2026-08-06T20:29:00.154021Z","title":"Prmbench: A fine-grained and challenging benchmark for process-level reward models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.154021Z"},"links":{"cited_paper":"/paper/2501.03124","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:52ef030c3d3c07f4716b0657ee37c78b9d5803a9cc882cfb35ba165c34356806","observation_id":"af15ffc1-4b55-4d7b-893f-55462028cbe3","resolution":{"observed_at":"2026-08-06T20:29:00.154021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-06T20:29:00.206673Z","title":"Training verifiers to solve math word problems, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.206673Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:b9709fc218031bdce61dc482810be44a0f4dd404d2ae5872a3d7e5001edaec17","observation_id":"242c2d42-a346-4983-a5f9-e9fb052d06a4","resolution":{"observed_at":"2026-08-06T20:29:00.206673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:05.606335Z","title":"Introducing gpt-4.1 in the api, Apr 2025","venue":null,"work_id":"f30196ea-2d3d-4c0b-9be8-609d76c9ffc5","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.270483Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:5611685863d7b444f280aad16d9874d86f577bb3e25b0351bbf8dd5b07396ce1","observation_id":"0a682beb-9d8b-4ffc-8185-1c377d0db050","resolution":{"observed_at":"2026-08-06T20:29:05.741704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:00.355018Z","title":"Let's verify step by step","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.355018Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:fdf1ca9a3cedb974c7a3960993517deb4be04666f6a2975d83423157fe284375","observation_id":"547769ab-98dd-42f4-8e5d-f44fecc985c4","resolution":{"observed_at":"2026-08-06T20:29:00.355018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:00.438344Z","title":"Measuring mathematical problem solving with the MATH dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.438344Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:7d2e73c8268fb81337e337a953d1df435824073409e69a8227a7f1adf3fe7084","observation_id":"cdac7ea2-3ff3-405a-9392-f1aa43da387b","resolution":{"observed_at":"2026-08-06T20:29:00.438344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:00.521903Z","title":"Transformers: State-of-the-art natural language processing","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.521903Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:583f4fe175f919d7063a0f5072e8f7ed43a5c217b374408402495210a9be98c6","observation_id":"d1a30e71-b262-458f-8efa-c396237e35b6","resolution":{"observed_at":"2026-08-06T20:29:00.521903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T20:29:00.633907Z","title":"Zhang, Han Bao, Hanwei Xu, Haocheng Wang, Haowei Zhang, Honghui Ding, Huajian Xin, Huazuo Gao, Hui Li, Hui Qu, J","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.633907Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:9c24638ef45d6d9dd804ad6431a6d82318a8a6211fa5f0a8faaf5e31ee6d23f1","observation_id":"98372d2d-19f9-45e5-8238-be050bbd90ed","resolution":{"observed_at":"2026-08-06T20:29:00.633907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-06T20:29:00.711530Z","title":"Qwen2.5 technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.711530Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:28c2bbd952a440ef56b1e8104084879afa2977187dc72ede24cc78c7f0a85a5b","observation_id":"5e52510a-4855-4949-9e1d-a50fd817f0f4","resolution":{"observed_at":"2026-08-06T20:29:00.711530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:05.285132Z","title":"Llama 3.3, Dec 2024","venue":null,"work_id":"560e69f3-3bad-4c9b-a900-f1bcb8599db2","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.790776Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:6015815c0f785f5cec7b0d74fff14a69b3ad7e7e1f02675b6c067016bd875fd5","observation_id":"a5b12f90-c1f5-4320-988b-69bde06197c9","resolution":{"observed_at":"2026-08-06T20:29:05.454473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.08905","last_updated":"2024-12-12T03:37:41Z","snapshot_observed_at":"2026-08-05T04:04:21.846023Z","submitted_at":"2024-12-12T03:37:41Z","title":"Phi-4 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.08905","snapshot_observed_at":"2026-08-06T20:29:00.903257Z","title":"Hewett, Mojan Javaheripi, Piero Kauffmann, James R","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.903257Z"},"links":{"cited_paper":"/paper/2412.08905","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:cba5e66af3400d27180fbdd1bb82914146723f584059d8a808dab0878b46c9d3","observation_id":"070d8f16-3cc9-4caa-ba86-afcde95ccfc4","resolution":{"observed_at":"2026-08-06T20:29:00.903257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-06T20:29:01.073515Z","title":"Qwen2 technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.073515Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:dcf50869ca61e6463a73a0d08a3ce8b968c891c5db531ebab93abb7a98dbce57","observation_id":"d8898eac-0793-4381-a9b0-2df54c6e3a79","resolution":{"observed_at":"2026-08-06T20:29:01.073515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T20:29:01.235737Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.235737Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:b87ef77785f1814fda432f783afba8fee9c03db0c53a8d54ad892008497befbd","observation_id":"598889cd-4a56-43db-8cd4-7e4e3b4c55bc","resolution":{"observed_at":"2026-08-06T20:29:01.235737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:04.911354Z","title":"Mistral small 3, Jan 2025","venue":null,"work_id":"a179b96e-d5d0-4075-9948-7e1d2bd6577e","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.386989Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:b2f570b5c5e35b7a4ddde0cd5175c53838360ff018485ec9a697a048ce10136c","observation_id":"9c5dc13c-19b8-4c72-8e21-1ba13f577e88","resolution":{"observed_at":"2026-08-06T20:29:05.122636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-06T20:29:01.501958Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.501958Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:87e9a4b31df1862bdd715e4b83570ee5e9e74d437f546d2e925970dbaecfe2c1","observation_id":"91a1453a-6c9b-4155-aebf-55fa2131cd9f","resolution":{"observed_at":"2026-08-06T20:29:01.501958Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:01.660462Z","title":"Impact of pretraining term frequencies on few-shot numerical reasoning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.660462Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:a7000c0d12e3cb0c0c8553c79a0d6bc11b72a17627e7a0de8ecf580d39ba4f08","observation_id":"8a78e84b-9451-4307-9a7a-d78784925c50","resolution":{"observed_at":"2026-08-06T20:29:01.660462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:04.670597Z","title":"Smith, Sarah Wiegreffe, and Yanai Elazar","venue":null,"work_id":"31a41ed1-6737-45ae-ac97-daf971976ed6","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.845245Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:175f10f64d0869cd4b2817d3d92f560983141d12bfed81ad9d3a86f011b55ff9","observation_id":"a8943ef9-1315-4cd0-8398-f439535fbb8e","resolution":{"observed_at":"2026-08-06T20:29:04.782194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:01.991836Z","title":"o pf, Yannic Kilcher, Dimitri von R \\","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.991836Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:3cbac4581bafe2dd31020d9d2b13f8e08cee14c483d51d555a36be7b3f0bd081","observation_id":"e5ca97be-2dab-4115-a7b7-171ee449086e","resolution":{"observed_at":"2026-08-06T20:29:01.991836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:02.140460Z","title":"Openhermes 2.5: An open dataset of synthetic data for generalist llm assistants, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.140460Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:bcdfd478c40daa2b99dd026c9ddaa4eabb55eda5564001f9801de76b285fe2de","observation_id":"9472acb6-eb08-412e-a62b-0d21af73a4f9","resolution":{"observed_at":"2026-08-06T20:29:02.140460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11116","last_updated":"2025-06-09T06:37:15Z","snapshot_observed_at":"2026-08-07T05:30:36.959314Z","submitted_at":"2025-06-09T06:37:15Z","title":"Infinity Instruct: Scaling Instruction Selection and Synthesis to Enhance Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11116","snapshot_observed_at":"2026-08-06T20:29:02.244807Z","title":"Infinity instruct: Scaling instruction selection and synthesis to enhance language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.244807Z"},"links":{"cited_paper":"/paper/2506.11116","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:ab0a7a339cf7f68af55fecf327fc1c234cdc44571b4e8246789c42dffd0628a6","observation_id":"c317eade-a7e4-48ad-8654-5687149260d8","resolution":{"observed_at":"2026-08-06T20:29:02.244807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:04.405551Z","title":"Ultrafeedback: Boosting language models with high-quality feedback, 2024","venue":null,"work_id":"63259950-d6ec-4139-bdd8-68e6fe16695a","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.356141Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:8b642f5fa10132bed30352fd4d62085dd747c4d662f20cb5393da41a8f853746","observation_id":"61ed6acb-e56a-49e0-8e0a-49c8c23c40b5","resolution":{"observed_at":"2026-08-06T20:29:04.501080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:02.487929Z","title":"Hwang, Jiangjiang Yang, Ronan Le Bras, Oyvind Tafjord, Christopher Wilhelm, Luca Soldaini, Noah A","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.487929Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:6c6d45c5b60e6a8be8f89de48ba0756feff983eb9dc2780eec0da67a963a7dd3","observation_id":"d42f5cd2-00f6-4eeb-98c0-831a16acdeb7","resolution":{"observed_at":"2026-08-06T20:29:02.487929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:02.599732Z","title":"Open r1: A fully open reproduction of deepseek-r1, January 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.599732Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:0e3995b6e16fb17c00285bc47334e92524969f436007c877d06042017e153c44","observation_id":"fa7a74bf-6fcc-4198-83eb-b128ade148e9","resolution":{"observed_at":"2026-08-06T20:29:02.599732Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.04178","last_updated":"2025-06-05T02:21:52Z","snapshot_observed_at":"2026-08-09T02:57:31.005115Z","submitted_at":"2025-06-04T17:25:39Z","title":"OpenThoughts: Data Recipes for Reasoning Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.04178","snapshot_observed_at":"2026-08-06T20:29:02.783626Z","title":"Merrill, Tatsunori Hashimoto, Yejin Choi, Jenia Jitsev, Reinhard Heckel, Maheswaran Sathiamoorthy, Alexandros G","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.783626Z"},"links":{"cited_paper":"/paper/2506.04178","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:5ad5393b1e2ba90fb239e82bdb1ff5d34cdaa328d6f9ab999ee7ad7c05bee7d0","observation_id":"fab9b386-0834-49e7-9db5-dc727f9131ea","resolution":{"observed_at":"2026-08-06T20:29:02.783626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:02.938680Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.938680Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:8acf97c07878117a99ee88a0d19576e41484a9169ca603e2a391f679c2a4df99","observation_id":"432e34ca-6d12-4311-9df2-c0977733d550","resolution":{"observed_at":"2026-08-06T20:29:02.938680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.20689","last_updated":"2024-03-29T07:17:39Z","snapshot_observed_at":"2026-08-07T10:31:33.290602Z","submitted_at":"2023-10-31T17:52:22Z","title":"Learning From Mistakes Makes LLM Better Reasoner","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.20689","snapshot_observed_at":"2026-08-06T20:29:03.095630Z","title":"Learning from mistakes makes llm better reasoner, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.095630Z"},"links":{"cited_paper":"/paper/2310.20689","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:b47a5b358cce596968accc4c7cf4957db900b3daf5519fa6271ab364260d348d","observation_id":"5a580e71-de61-4a51-819e-81f6b955611e","resolution":{"observed_at":"2026-08-06T20:29:03.095630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17703","last_updated":"2025-03-29T15:21:55Z","snapshot_observed_at":"2026-08-10T04:31:53.577547Z","submitted_at":"2025-01-29T15:20:30Z","title":"Critique Fine-Tuning: Learning to Critique is More Effective than Learning to Imitate","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17703","snapshot_observed_at":"2026-08-06T20:29:03.216236Z","title":"Critique fine-tuning: Learning to critique is more effective than learning to imitate, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.216236Z"},"links":{"cited_paper":"/paper/2501.17703","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:bc205023183afe91ff8a90fd48eb22a329db0b89591324c1f8c0edd2cddfbf92","observation_id":"f256587f-fd18-4918-be5b-1e797c5c666e","resolution":{"observed_at":"2026-08-06T20:29:03.216236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:03.409054Z","title":"The effect of sampling temperature on problem solving in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.409054Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:f206f804b1cd71652a3b7bc976563afb040e825fbbeda038a4833a3a6188d8fa","observation_id":"4eb03d92-b6a5-4658-90b2-9dd56e085279","resolution":{"observed_at":"2026-08-06T20:29:03.409054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:03.530733Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.530733Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:dc5c8a436a28e35a625fca63e7d982848c12911350205357bb87cab363871d18","observation_id":"d70d4393-6219-408a-8892-8d96da99a79c","resolution":{"observed_at":"2026-08-06T20:29:03.530733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:03.675873Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.675873Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:cca43345fdb61d5af0e46fd7e767b7c5dba17be7ea71276bc56b8ee3ecea5ef0","observation_id":"a4614c9e-01f7-4332-adb6-eec84c16fca5","resolution":{"observed_at":"2026-08-06T20:29:03.675873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:03.805548Z","title":"after incorrect reasoning or answer to prompt LLMs to self-correct, without finetuning. We observe significant reductions in the blind spot after appending ``Wait","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.805548Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:5162b13d599436fe1d6c8ad78ef061583965b5926a41251bb250ec7e86ea98a1","observation_id":"e8dda0b1-32fe-40b6-bebf-104c5fd66015","resolution":{"observed_at":"2026-08-06T20:29:03.805548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T09:35:34.594850Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":45,"verified_exact":0,"verified_fuzzy":15},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 4 inbound Pith citation observations for arXiv:2507.02778."}