{"as_of":"2026-08-10T08:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:94d69dfd582b7c9bf6cd057086f68eb374425fb0e4fa866d0b13036753caa927","coverage":[{"denominator":108,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T03:56:28.073779Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.13753/citation-record","integrity":"/paper/2607.13753/integrity","json":"/paper/2607.13753/citation-record.json","paper":"/paper/2607.13753"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:16.720764Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:16.720764Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:1573e4d3b9dadccddeadf61b986c19f5785f71ebc43423d0c3c6cb76aaf1ba4b","observation_id":"443ca3e7-4616-46c7-96cb-be21c0a9fc7c","resolution":{"observed_at":"2026-08-02T03:56:16.720764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:16.883058Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:16.883058Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:856c5981d6e24e55c0ea51d8b88dfb1a1c2f679e5b51855550ae8974c262c4fc","observation_id":"8137768e-02ec-4d3a-afb5-67e47d2f671f","resolution":{"observed_at":"2026-08-02T03:56:16.883058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.046894Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.046894Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:7cfaab168d487d7633eefa7d3d8e23d6a3174816fd0be19e38a62234592cb064","observation_id":"b818d33e-8ba0-4a99-ba8b-dd3b1a83e111","resolution":{"observed_at":"2026-08-02T03:56:17.046894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.305342Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.305342Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:ea76c616e977d220ed280320ba286a7a6a86b4f13c135d895bcb1a5c864368a3","observation_id":"995417c0-b4b1-426d-a748-ff81dcc1857a","resolution":{"observed_at":"2026-08-02T03:56:17.305342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.525161Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.525161Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:d73aeccf5a9c651219f2f6e956e507c1322322bcf036fcfdacb3d74be35d4393","observation_id":"4ee00f7c-e497-4795-8696-7bf719da8121","resolution":{"observed_at":"2026-08-02T03:56:17.525161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.635578Z","title":"2021 , eprint=","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.635578Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:1bd83d128fb5e022f738cc8598b29ab3c3910b3fd851dd0ec8acecacd78665f1","observation_id":"303f3ccf-1834-4265-9893-814d146fa234","resolution":{"observed_at":"2026-08-02T03:56:17.635578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.746809Z","title":"2022 , eprint=","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.746809Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:323c1bb857e9ea3164700e5307e16c35d058d905bffe28ff5016697e15c4476c","observation_id":"9bbd247a-6fae-4182-af5a-2a9e713390b5","resolution":{"observed_at":"2026-08-02T03:56:17.746809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.858651Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.858651Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:b7b82ef6968e8db14f3e441a403453af26b5d44c00fa84f730b9456f58cacbc0","observation_id":"481f8c39-189d-4581-8f4e-39db76134d38","resolution":{"observed_at":"2026-08-02T03:56:17.858651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.970359Z","title":"Findings of the Association for Computational Linguistics: ACL 2025 , pages=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.970359Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:2b5a13bebab0a9a09f0dcd9085940eb26018338fcc4d9730bb3eeed46429203b","observation_id":"a5aedc63-fff2-49b5-82b3-d7e73f61193d","resolution":{"observed_at":"2026-08-02T03:56:17.970359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.081629Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.081629Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:d6fa1a61a1a5b4fceebd0fe84b90264102e90b40bb3eff488c735eb07e007bb5","observation_id":"8db46a12-1db6-41bd-9b33-2135fa964d2b","resolution":{"observed_at":"2026-08-02T03:56:18.081629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.155745Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.155745Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:bd8d2c150bcde1d4fa91ecbc9eeb3f563b233c333e28b8e8257c2dfe6e8b1df9","observation_id":"14a5ed35-9b7e-417b-9ebc-e76c0a4a07e8","resolution":{"observed_at":"2026-08-02T03:56:18.155745Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.269368Z","title":"International conference on machine learning , pages=","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.269368Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c04ade5b32a99b378181df92ca591d62d03721e39dcdc7f1d89ed3ed7dc946d6","observation_id":"d5eaecc8-914e-4465-a95f-2b7cfeb643e1","resolution":{"observed_at":"2026-08-02T03:56:18.269368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.344443Z","title":"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers) , pages=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.344443Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:167f442b65aa986cf3526021cef613e5bc124742b8395130d35b875a1eee1259","observation_id":"70570ddf-5a85-4d37-877f-27f3bf62b6ed","resolution":{"observed_at":"2026-08-02T03:56:18.344443Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.454309Z","title":"Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.454309Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:3dfdfd310edf087cb2c0be00e1b742957e871bd2946146c024d2c335efcaee10","observation_id":"6db761c1-bac1-4316-9bb1-20e4ce62ceb3","resolution":{"observed_at":"2026-08-02T03:56:18.454309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.671429Z","title":"2022 , eprint=","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.671429Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:52e02b49b5d9d8f3d0f85335efdaaecdf572bf4119e0c7a65114c912a6a76c83","observation_id":"ae9d9765-a8e3-4738-9e46-66826995793e","resolution":{"observed_at":"2026-08-02T03:56:18.671429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.942951Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.942951Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:5c9833eb35039f8e17ec8193f0c7eb698e911645bb1dfa0fecb3d40ae3d1afac","observation_id":"1553f37e-77f9-475d-a7cd-62e039ee26cc","resolution":{"observed_at":"2026-08-02T03:56:18.942951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.030784Z","title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.030784Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:a8d40cec3219a98c75fe8aff64b3b4e6a06884fa1c7c06f23e956246a547bd92","observation_id":"35799c2f-73dd-4fcd-957d-f4d650b189a9","resolution":{"observed_at":"2026-08-02T03:56:19.030784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.161410Z","title":"and Hauskrecht, Milos , title =","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.161410Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:63fd15341446834262231d1a236c4b19e555d8a776b65f3490810c5149da4252","observation_id":"0d752492-ea31-4e86-a8d5-2bb895aa4e13","resolution":{"observed_at":"2026-08-02T03:56:19.161410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.236437Z","title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP) , pages=","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.236437Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:ffca4f71dc12877fadb2be75c695208459a0df102db3f9f8a0cf10a0fbb13f16","observation_id":"bfa3907a-f566-461b-bf16-2ecafcc4d116","resolution":{"observed_at":"2026-08-02T03:56:19.236437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.311265Z","title":"Transactions of the Association for Computational Linguistics , volume=","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.311265Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:dca8f4250e5cfffdac3fda899e3b9a6bc4f9184fcd907b6dca4496c5ee4b800a","observation_id":"2d482648-1523-4db9-949e-092c661cb44d","resolution":{"observed_at":"2026-08-02T03:56:19.311265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.357682Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.357682Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:2f4a051b6b76495c61f2c868737d2ff31038a87d4c756f24771d7dffaf4186a9","observation_id":"de624713-9517-46d4-8228-da746102f819","resolution":{"observed_at":"2026-08-02T03:56:19.357682Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.405028Z","title":"Probabilistic Outputs for Support Vector Machines and Comparisons to Regularized Likelihood Methods , volume =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.405028Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:91b98c1097d7e53c15b0ed224d898db16ccc275eeb04bda598aa06ef42f82a1d","observation_id":"5872479e-bfa7-4646-9170-0f548b7c7fc8","resolution":{"observed_at":"2026-08-02T03:56:19.405028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.467260Z","title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.467260Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:955bd8b2a2300b3e06e3f5c841de8f8e68b2d378cf4037c4d08f1702a3d211cd","observation_id":"2931a62b-0358-4aef-a985-a22bbc4e8700","resolution":{"observed_at":"2026-08-02T03:56:19.467260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.539479Z","title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.539479Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:cc3c176e0ce8ec2a96704c198138a9b23931586e2570619bfae7f4945356cdf1","observation_id":"1d21ee5c-4e73-461c-9dd0-3a159e64e037","resolution":{"observed_at":"2026-08-02T03:56:19.539479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.688306Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.688306Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:125fd3d5e9f68502b27c4baecd775e3ad0b09a3abba159a0846187c88f0bd557","observation_id":"14429ecd-e4fc-40fb-bdd2-dcf829f58a32","resolution":{"observed_at":"2026-08-02T03:56:19.688306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.785245Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.785245Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:fa275f21ffd812a1e3cc45993d9b92debc2ae0c0316a383ec3f71644af130ca1","observation_id":"b8358bdf-efb0-4455-9865-cf423c91d429","resolution":{"observed_at":"2026-08-02T03:56:19.785245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.841972Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.841972Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c7747cfe004693c8718af9f269a256d52ebfb7e676d1f348ce0f698d2115dae7","observation_id":"a8f3c383-7b93-48bf-8d1b-36fbef7d563e","resolution":{"observed_at":"2026-08-02T03:56:19.841972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.896437Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.896437Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:9f8baf123f15846e1a0401eeddb433db1eea96aa89388ca14014e11988e5018c","observation_id":"3cfe850b-d513-4f74-ba7d-7fe46af8b78d","resolution":{"observed_at":"2026-08-02T03:56:19.896437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.967258Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.967258Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:758a912cee4ac0dff94d20b13d27b9538556029d0bed917fef89b290c8351dbb","observation_id":"2ec03989-89e6-4858-bd76-777660b7ef7a","resolution":{"observed_at":"2026-08-02T03:56:19.967258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:20.024839Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.024839Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:327748a48842fe81b50b9dd3c8da611ae4db3257e0dcd8076f08dea03b654aaf","observation_id":"23f17299-ffe3-49a9-b8d2-093aca93aa4a","resolution":{"observed_at":"2026-08-02T03:56:20.024839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-02T03:56:20.086170Z","title":"arXiv preprint arXiv:2501.12948 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.086170Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:bc04e604970bec0a21cb8bb15f5166f3d7ed1139f55acd8bc6c3a1afad1409ce","observation_id":"d0caa55b-9345-4fce-b06c-cf68ef045cb4","resolution":{"observed_at":"2026-08-02T03:56:20.086170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:20.204278Z","title":"2025 , eprint=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.204278Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:4bc0746bf7431e02a3386d61ded53f0f0461b5edf1e6a9e2716effbfb5807b08","observation_id":"bbb41a7b-cb88-49c8-b141-d2a90c8abb8e","resolution":{"observed_at":"2026-08-02T03:56:20.204278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:20.271780Z","title":"2025 , eprint=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.271780Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:64c9722d1d4e9ff5339a68d0fdb284d96431917154ff5effbf8fe802addf8737","observation_id":"ba77f5a3-39ae-42c2-9381-757c04c03ad0","resolution":{"observed_at":"2026-08-02T03:56:20.271780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12186","last_updated":"2024-11-12T13:24:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:57:57Z","title":"Qwen2.5-Coder Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12186","snapshot_observed_at":"2026-08-02T03:56:20.714439Z","title":"5-coder technical report , author=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.714439Z"},"links":{"cited_paper":"/paper/2409.12186","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:8c0a2881eda3452509dc1ae00a066b48759803a9587f4745ce9150fa2e0269b7","observation_id":"f449d21f-b2be-439f-b653-db349c7fd0fe","resolution":{"observed_at":"2026-08-02T03:56:20.714439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:20.799675Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.799675Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:ddc83035d5cb2990c3832f72c9d706f89c7c38934b748ea6a55eae1565b2d515","observation_id":"896c379b-cc53-476c-90ac-d0aef515b035","resolution":{"observed_at":"2026-08-02T03:56:20.799675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:20.880149Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.880149Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:1c81fa2cdc37747fea9c058107faa78a2b46215029f07529603f9462b4897f98","observation_id":"3de3af29-3ee5-4466-a8be-a08728f475c2","resolution":{"observed_at":"2026-08-02T03:56:20.880149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:21.053384Z","title":"2025 , eprint=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:21.053384Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:6cbc112f7c7976cd95b56b661b1db5821adf6f87fe3bc34438f31e6edee2da61","observation_id":"ce50c94b-a3a4-471f-be24-24441214db39","resolution":{"observed_at":"2026-08-02T03:56:21.053384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-02T03:56:21.137234Z","title":"arXiv preprint arXiv:2307.09288 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:21.137234Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:80ec78c4940603ef09d545a041ea40b0afbf729038229bf2d29541b394df0da7","observation_id":"7c694cdb-d3f0-43dd-8de0-b2ad25436cbf","resolution":{"observed_at":"2026-08-02T03:56:21.137234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:21.226495Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:21.226495Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:7e18b0a21a790d91282cdba18531d833755dffac04a4aee5f197908155aae1b4","observation_id":"a7dfab01-8fb4-4bbf-80ed-fc97908c1d1b","resolution":{"observed_at":"2026-08-02T03:56:21.226495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:21.778799Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:21.778799Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:89545f645723777c5286a85d3f993268881bf8c525e1ae58068bd353fd5c9c3b","observation_id":"5dcdc66f-3e2e-4473-8687-5f67dabb6097","resolution":{"observed_at":"2026-08-02T03:56:21.778799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:21.914841Z","title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:21.914841Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:7c2de9b51a75610463b873280a70251ecafd74931d886eb224a765990347842d","observation_id":"0df156b7-8300-426f-a02a-3068a2c613f2","resolution":{"observed_at":"2026-08-02T03:56:21.914841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:22.005298Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.005298Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:ec959aa9e75e9e5079c649586d97858ed384ce2e026ac16b5ff0d02847eb3506","observation_id":"4a335679-0cce-4c6c-b368-e2629f9fe916","resolution":{"observed_at":"2026-08-02T03:56:22.005298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:22.088004Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.088004Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:d59274ff346b028db371d07638bc628b46eccaecd50d0aef72051b080001a7be","observation_id":"b80a5562-e38f-49c0-bb48-a47b19172bb9","resolution":{"observed_at":"2026-08-02T03:56:22.088004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:22.483686Z","title":"2021 , eprint=","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.483686Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:3dcc184bbf858618c84402d32321b8d9c7ff2e9d25d1a8034a4729e48d800f93","observation_id":"81448863-aa35-4d29-8385-7335affaafb8","resolution":{"observed_at":"2026-08-02T03:56:22.483686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-08-08T11:58:24.516369Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-02T03:56:22.639590Z","title":"arXiv preprint arXiv:2107.03374 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.639590Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:da1f6ec118f623d6fc5ba5eb1b5877dcd7ed7173c94d88823215e24a1ac3ee48","observation_id":"ca59b8ff-fdf7-4965-84ec-c4351306bedb","resolution":{"observed_at":"2026-08-02T03:56:22.639590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:22.760061Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.760061Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:23114e0323ac86a95c25a5700b1e84e09a4cbfd91f64a20684a80afd745dc479","observation_id":"61fb4327-a01c-4cd3-a077-1fa48da83bf6","resolution":{"observed_at":"2026-08-02T03:56:22.760061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:22.894503Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.894503Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:8d1b0a065ab7da0b3028519e6a4ad38d733dec5217a85a8cc89eec632277f73f","observation_id":"d3b39df6-4f93-4661-9ee9-3888b3ca71ef","resolution":{"observed_at":"2026-08-02T03:56:22.894503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.030533Z","title":"2025 , note=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.030533Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:891b90eb43de3c415bb9231e17c91f787ba57761802b20521bbbbe43901d2ace","observation_id":"348de46d-1511-4261-bede-1369d2b10dc5","resolution":{"observed_at":"2026-08-02T03:56:23.030533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.120640Z","title":"Notion Blog , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.120640Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:bd7e148b813224c227211b1ef62122919c99c4353e75307ed37b1a95273a2001","observation_id":"f888b210-f776-43c4-b9c2-38565231cac3","resolution":{"observed_at":"2026-08-02T03:56:23.120640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.165656Z","title":"Proceedings of the Twentieth European Conference on Computer Systems , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.165656Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:5ad7714cd27369f95ae86c3b6f362278725bdaa17ebebc5ceceb56dbbddbce58","observation_id":"83eb78a5-a572-4d8c-8c7a-c698a76b7835","resolution":{"observed_at":"2026-08-02T03:56:23.165656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21787","last_updated":"2024-12-30T19:03:24Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:57:25Z","title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21787","snapshot_observed_at":"2026-08-02T03:56:23.249870Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.249870Z"},"links":{"cited_paper":"/paper/2407.21787","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:a57fd660eef8b77086a94706256f4e9d3c365642402cd259bae39fd276429b0d","observation_id":"67ad94e3-5616-4e9f-917a-d7a4dcb50dd3","resolution":{"observed_at":"2026-08-02T03:56:23.249870Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.361339Z","title":"Suchanek, and Gaël Varoquaux","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.361339Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:695a91a5a2f5e4bbe3d7eaef740120806b39b954417e112638a14a5669af41de","observation_id":"d5a324d4-d99d-4d29-b747-38185b5d4cd5","resolution":{"observed_at":"2026-08-02T03:56:23.361339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.440593Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.440593Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:23e0d529d65fdebf85a78bfdc8b4b5b8f53d29743ada7adb6840ff20b266caba","observation_id":"3c453e4c-d9c0-4800-880b-d144d49c2f3d","resolution":{"observed_at":"2026-08-02T03:56:23.440593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-02T03:56:23.497995Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.497995Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:201ccb31566016dab412c4578c6cbbfd90cc907ba9a0fdecaddb5a8c75a29c64","observation_id":"4ef649a2-8b45-4b4d-8810-387b6f215161","resolution":{"observed_at":"2026-08-02T03:56:23.497995Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.548141Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.548141Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:34479130c9aed008e11e38136d4fcf77fc41aa43ea90b715077c1a8dfe64977e","observation_id":"ac7bb018-069e-414f-a716-39959d97cc90","resolution":{"observed_at":"2026-08-02T03:56:23.548141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.15260","last_updated":"2025-08-21T05:48:38Z","snapshot_observed_at":"2026-08-02T03:10:09.020351Z","submitted_at":"2025-08-21T05:48:38Z","title":"Deep Think with Confidence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.15260","snapshot_observed_at":"2026-08-02T03:56:23.603454Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.603454Z"},"links":{"cited_paper":"/paper/2508.15260","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:5c8982a1cc103b28180e2c1d6f71e34a1a86feced3bdfc1b55bc6fbe8d5ee211","observation_id":"feb18894-f6f1-442e-921f-1624600d7218","resolution":{"observed_at":"2026-08-02T03:56:23.603454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.658276Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.658276Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:cfeaa13cbe8065f8a4482f186be2251d667c098bf9a3426d0f83f4f0b15f1f86","observation_id":"3ee690ed-577e-4134-96b9-af4d7d7f4621","resolution":{"observed_at":"2026-08-02T03:56:23.658276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.714297Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.714297Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:59c0e492c5c25b83664bd9461008b181d36d35914a96f54fa18f8c596784efed","observation_id":"b82e8989-5a63-4900-8338-5ac5bd326d37","resolution":{"observed_at":"2026-08-02T03:56:23.714297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.768342Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.768342Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:41ed059c10c07b848fc401f8fad6dae1b71d57eeca10aa8a5744f454e711c52e","observation_id":"961b300a-ac3f-40e0-908f-beb82197fc5c","resolution":{"observed_at":"2026-08-02T03:56:23.768342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.823431Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.823431Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:9819a784765d6c9341e6b2fb1944c9e16845df0dd5db14a5f147d67e8f867fdc","observation_id":"fa844820-789c-4b45-98c1-a39b92167c1a","resolution":{"observed_at":"2026-08-02T03:56:23.823431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-02T03:56:23.888612Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.888612Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:d5729cdbb209453ad69ec4511b3917623cf70dce037b010d4876fecfbca4ae13","observation_id":"21d58873-741b-4fd8-a84f-6f553481f917","resolution":{"observed_at":"2026-08-02T03:56:23.888612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1503.02531","last_updated":"2015-03-09T15:44:49Z","snapshot_observed_at":"2026-07-06T04:11:24.157003Z","submitted_at":"2015-03-09T15:44:49Z","title":"Distilling the Knowledge in a Neural Network","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1503.02531","snapshot_observed_at":"2026-08-02T03:56:23.929756Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.929756Z"},"links":{"cited_paper":"/paper/1503.02531","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:d984c47c6992d06a3cebdb741e0e8ff3ce45a8e2f4c8bd6bf50a757291dbd3f0","observation_id":"799011ba-2eaa-4488-894b-d1c81aebb9ee","resolution":{"observed_at":"2026-08-02T03:56:23.929756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.06457","last_updated":"2024-08-14T02:41:48Z","snapshot_observed_at":"2026-07-06T17:28:02.037844Z","submitted_at":"2024-02-09T15:02:56Z","title":"V-STaR: Training Verifiers for Self-Taught Reasoners","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.06457","snapshot_observed_at":"2026-08-02T03:56:24.014273Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.014273Z"},"links":{"cited_paper":"/paper/2402.06457","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c422daf15203e160be0647291ea15590fc62664d1bd557281f6b3f5aed9cc299","observation_id":"0ff3e0b5-87f7-449e-b92d-40ba825e54a2","resolution":{"observed_at":"2026-08-02T03:56:24.014273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:24.068987Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.068987Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:171fd24ef60d34664ca8afa169026fcec5d992c21a2a606e09a50dedb20d9585","observation_id":"75be82ed-5f57-4301-9cb3-29cd980988fa","resolution":{"observed_at":"2026-08-02T03:56:24.068987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.07079","last_updated":"2026-05-22T17:34:18Z","snapshot_observed_at":"2026-07-06T22:48:13.053749Z","submitted_at":"2026-03-07T07:26:18Z","title":"Entropy-Aware On-Policy Distillation of Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.07079","snapshot_observed_at":"2026-08-02T03:56:24.076911Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.076911Z"},"links":{"cited_paper":"/paper/2603.07079","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:8b60c8dcc22bfa52e21e28f402c3fabf39bff7a98fe939bb06dd217748473e13","observation_id":"2f972ddc-0bc2-405e-989b-8f7482a4d691","resolution":{"observed_at":"2026-08-02T03:56:24.076911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:24.096298Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.096298Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:31c26518185e31d1da30c02a58c522cce78e3f6bc5336f0109d6d6d39cdc19eb","observation_id":"03eea6e4-c882-4cbd-872f-e4a7e795d3a8","resolution":{"observed_at":"2026-08-02T03:56:24.096298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.05221","last_updated":"2022-11-21T16:38:35Z","snapshot_observed_at":"2026-08-06T08:34:11.887259Z","submitted_at":"2022-07-11T22:59:39Z","title":"Language Models (Mostly) Know What They Know","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.05221","snapshot_observed_at":"2026-08-02T03:56:24.247165Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.247165Z"},"links":{"cited_paper":"/paper/2207.05221","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c0597dbb25854330d0be98a1125ce76b0acc9d3f379b4b79f79f955ccf876d3a","observation_id":"554be546-d273-4163-b323-fd0cd6788333","resolution":{"observed_at":"2026-08-02T03:56:24.247165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:24.403912Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.403912Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:b3b9af502f25f948399d134f4dc8007b66a052da41d111ce8f0b34039f2217fd","observation_id":"afd62219-3532-4b9e-887e-6aa5939bf3d1","resolution":{"observed_at":"2026-08-02T03:56:24.403912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.09664","last_updated":"2023-04-15T12:55:45Z","snapshot_observed_at":"2026-07-06T14:53:27.667483Z","submitted_at":"2023-02-19T20:10:07Z","title":"Semantic Uncertainty: Linguistic Invariances for Uncertainty Estimation in Natural Language Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.09664","snapshot_observed_at":"2026-08-02T03:56:24.521676Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.521676Z"},"links":{"cited_paper":"/paper/2302.09664","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:cd2f80a8ff53fdb57600e6d317729709c7cb344e69e9f9fb471a5e6006829db9","observation_id":"680bbf43-3e73-46f7-a35d-adc395118d28","resolution":{"observed_at":"2026-08-02T03:56:24.521676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.14858","last_updated":"2022-07-01T02:15:12Z","snapshot_observed_at":"2026-08-05T15:41:22.691461Z","submitted_at":"2022-06-29T18:54:49Z","title":"Solving Quantitative Reasoning Problems with Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.14858","snapshot_observed_at":"2026-08-02T03:56:24.673658Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.673658Z"},"links":{"cited_paper":"/paper/2206.14858","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:04c5fd95ddfa7495d494117574f8e3b765cf3a681d7682c234cf4bd56a8119d6","observation_id":"c8716189-2874-4206-8d3d-81b5277bf5fa","resolution":{"observed_at":"2026-08-02T03:56:24.673658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:24.777752Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.777752Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:d57309d24280200a17869e1bc0d96c7eb15118a2605abd70c135319c181ec6f6","observation_id":"3920efc6-6911-4955-a8a4-d4ed1b692a53","resolution":{"observed_at":"2026-08-02T03:56:24.777752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.14334","last_updated":"2022-06-13T05:04:53Z","snapshot_observed_at":"2026-08-09T23:45:38.167954Z","submitted_at":"2022-05-28T05:02:31Z","title":"Teaching Models to Express Their Uncertainty in Words","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.14334","snapshot_observed_at":"2026-08-02T03:56:24.978542Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.978542Z"},"links":{"cited_paper":"/paper/2205.14334","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c1735615822ba0754605f28ac396fc88b18c22ddd3b922d72f832d742326e7db","observation_id":"8dc3cc09-663e-4bd5-aaf4-9188b42317c6","resolution":{"observed_at":"2026-08-02T03:56:24.978542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:25.128541Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.128541Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:46e949e77eec694b034f25b63bbaeea7daacd0eaea4ad26a4490ce26130f0aef","observation_id":"d63560fb-7b10-4f43-baa7-64685e72286a","resolution":{"observed_at":"2026-08-02T03:56:25.128541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:25.283578Z","title":"Tang, Manan Roongta, Colin Cai, Jeffrey Luo, Tianjun Zhang, Li Erran Li, Raluca Ada Popa, and Ion Stoica","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.283578Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:ee135c2f2de54a82ca3a08f730ebfb51b4e711543e27b09ce6d13c7e88164a13","observation_id":"4b0f8d4a-8e97-48d2-bd37-9c2575b34c86","resolution":{"observed_at":"2026-08-02T03:56:25.283578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:25.392473Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.392473Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:b1a30d62751531d9308250583ab84f54a5669d84d7eaeb669bacae8f436d1041","observation_id":"823f7f7e-bf7b-4604-8f60-852d7fc3c112","resolution":{"observed_at":"2026-08-02T03:56:25.392473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:25.502679Z","title":"Cooper, and Milos Hauskrecht","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.502679Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:0915092bd6f45aa9fdf10af93074f88fef6fc7deb519a90fc01268d5c9c143ea","observation_id":"0cb65d26-ae1a-4f91-a47f-594dcfa87961","resolution":{"observed_at":"2026-08-02T03:56:25.502679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.00114","last_updated":"2021-11-30T21:32:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-11-30T21:32:46Z","title":"Show Your Work: Scratchpads for Intermediate Computation with Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.00114","snapshot_observed_at":"2026-08-02T03:56:25.601371Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.601371Z"},"links":{"cited_paper":"/paper/2112.00114","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c3eb22243f1d3b6483f60feb261794adc257227f62d5f8de16209f3100b8de50","observation_id":"0b77afcd-db0d-4e7f-8d39-71960060d9f8","resolution":{"observed_at":"2026-08-02T03:56:25.601371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-02T03:56:25.765151Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.765151Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:d917319f32a7bcc91f72a5db7ad7d0161ff252edabcb45a8dab622b2e2137d00","observation_id":"ee3d3407-7d5b-4cf3-9a81-0fbaf9356a30","resolution":{"observed_at":"2026-08-02T03:56:25.765151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:25.929517Z","title":null,"venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.929517Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:f5d577df8214493ceb9091034f3aadfb745e34ca1f06d7a95be99d3f8ecbcfe4","observation_id":"8dcf0309-6377-4327-b1e6-74cf0cf49094","resolution":{"observed_at":"2026-08-02T03:56:25.929517Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-02T03:56:26.085330Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.085330Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:39b661365bfecedd55af1634d77b72fc3938229d5da65d66153aea2ca641ec32","observation_id":"23048bec-69ee-42ce-ae5d-6f87b4149b89","resolution":{"observed_at":"2026-08-02T03:56:26.085330Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-02T03:56:26.201200Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.201200Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:af1730dc558dbb46849fc371ab7107ef15e322781e9542131110200a31dd66ff","observation_id":"cff3b64b-e71c-4d53-a4e5-2c8f3aed78e4","resolution":{"observed_at":"2026-08-02T03:56:26.201200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-02T03:56:26.300681Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.300681Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:fd2c5e8d6e21ca382a2cdc99948869657af5536fc893744ffe4b27fd01661c9e","observation_id":"baca7525-b88d-4f3b-868d-4c077fe4f304","resolution":{"observed_at":"2026-08-02T03:56:26.300681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:26.472307Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.472307Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:e9f167b1b3b53eb451d1b80c6b79add49551101a211b83fc994229bb0da09f8c","observation_id":"77d4108a-1dbd-40c2-9bc3-66605ce49084","resolution":{"observed_at":"2026-08-02T03:56:26.472307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06585","last_updated":"2024-04-18T03:12:09Z","snapshot_observed_at":"2026-08-08T12:03:18.710098Z","submitted_at":"2023-12-11T18:17:43Z","title":"Beyond Human Data: Scaling Self-Training for Problem-Solving with Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06585","snapshot_observed_at":"2026-08-02T03:56:26.701252Z","title":"Co-Reyes, Rishabh Agarwal, Ankesh Anand, Piyush Patil, Xavier Garcia, Peter J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":106,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.701252Z"},"links":{"cited_paper":"/paper/2312.06585","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c282e554d89c33a4d7d80d49e384eac61af1ccde9e9f46464b868157f99f9caa","observation_id":"e0ff153d-3cab-4e56-aead-9b7e7bd48a79","resolution":{"observed_at":"2026-08-02T03:56:26.701252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-02T03:56:26.864154Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":107,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.864154Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c93103d61fbe10fc5bc25a8c101a2ad1ec6fb72230c30d9de3e578973e610612","observation_id":"9c2de194-814c-47ec-aafd-045bfe1db7d1","resolution":{"observed_at":"2026-08-02T03:56:26.864154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:26.959184Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.959184Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:db499f48681841063f0d25f2d007189943f7a08c728ba19da102e8afcccd1207","observation_id":"dd6fe679-a26e-42ef-b5d6-8390dda94ef1","resolution":{"observed_at":"2026-08-02T03:56:26.959184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:27.043472Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":109,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.043472Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:7d6226f254a3353ab7775ade49d97acace7ef56ed0dcf0cb7fc12f2937707f33","observation_id":"bfe0b2c0-1b95-40de-b0f3-e1591a58c3a4","resolution":{"observed_at":"2026-08-02T03:56:27.043472Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.14275","last_updated":"2022-11-25T18:19:44Z","snapshot_observed_at":"2026-08-01T02:16:43.109337Z","submitted_at":"2022-11-25T18:19:44Z","title":"Solving math word problems with process- and outcome-based feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.14275","snapshot_observed_at":"2026-08-02T03:56:27.103545Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":110,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.103545Z"},"links":{"cited_paper":"/paper/2211.14275","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:bfcba12b839c500618b4f243f0f4320bc31a1114e00c6fd868fe48021c7c1dce","observation_id":"b3d05b55-f6b1-4e03-b256-80f2afb2143a","resolution":{"observed_at":"2026-08-02T03:56:27.103545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.09265","last_updated":"2025-09-11T08:50:01Z","snapshot_observed_at":"2026-08-08T10:27:53.369430Z","submitted_at":"2025-09-11T08:50:01Z","title":"Harnessing Uncertainty: Entropy-Modulated Policy Gradients for Long-Horizon LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.09265","snapshot_observed_at":"2026-08-02T03:56:27.159465Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.159465Z"},"links":{"cited_paper":"/paper/2509.09265","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:57c5dbed73e6d942d6caa60cdf2e831cd5fbd6062c34e47758c3752468154970","observation_id":"42bbac1a-559f-4d9c-b285-f70dfc65cc5b","resolution":{"observed_at":"2026-08-02T03:56:27.159465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:27.235699Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":112,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.235699Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:0103e67a84e9c64448e8a0d1b24f70f2eb6659407008c87b4991fae1792980c4","observation_id":"18fd48cc-3fa1-4ebe-ba9b-d3f7a56333c9","resolution":{"observed_at":"2026-08-02T03:56:27.235699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01939","last_updated":"2025-11-13T10:08:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-02T17:54:39Z","title":"Beyond the 80/20 Rule: High-Entropy Minority Tokens Drive Effective Reinforcement Learning for LLM Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.01939","snapshot_observed_at":"2026-08-02T03:56:27.321716Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":113,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.321716Z"},"links":{"cited_paper":"/paper/2506.01939","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:56d6b347be7e3652bca962ae4eeb4438d98628cd4187d14b6949f83d42e77e45","observation_id":"1e9aa211-3d85-4e7e-b7ee-9e22139d5307","resolution":{"observed_at":"2026-08-02T03:56:27.321716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-02T03:56:27.406144Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":114,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.406144Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:bb0f436d32bd1297e42da4e6eb9a2ea8d9ae7e4c341f5c1d50b108041b95bda2","observation_id":"aee81d70-fd58-468c-b624-0423198e0d1b","resolution":{"observed_at":"2026-08-02T03:56:27.406144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.00487","last_updated":"2026-05-30T02:39:40Z","snapshot_observed_at":"2026-08-02T08:25:04.138756Z","submitted_at":"2026-05-30T02:39:40Z","title":"TAPS: Target-Aware Prefix Tree Selection for Diffusion-Drafted Speculative Decoding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.00487","snapshot_observed_at":"2026-08-02T03:56:27.461043Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":115,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.461043Z"},"links":{"cited_paper":"/paper/2606.00487","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:08c5a3ed5b82466094758719b230659438a979bb01cf07c13f0f1750fdf4c715","observation_id":"895e38b0-66f5-49e2-89de-35a5e31d6882","resolution":{"observed_at":"2026-08-02T03:56:27.461043Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11903","last_updated":"2023-01-10T23:07:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-28T02:33:07Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.11903","snapshot_observed_at":"2026-08-02T03:56:27.545672Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.545672Z"},"links":{"cited_paper":"/paper/2201.11903","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:1ad7fb1bb1716e8a8dc7a58474ddf4068a350f891b138616a661104dc164f8ce","observation_id":"f6b27116-e6a6-4ab6-8f5c-b9a11e92cbd0","resolution":{"observed_at":"2026-08-02T03:56:27.545672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00724","last_updated":"2025-03-03T07:53:32Z","snapshot_observed_at":"2026-08-09T23:53:42.648697Z","submitted_at":"2024-08-01T17:16:04Z","title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00724","snapshot_observed_at":"2026-08-02T03:56:27.628364Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":117,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.628364Z"},"links":{"cited_paper":"/paper/2408.00724","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:7998afaf0e12b9401e418b412915312a4a957316a4e07e5252b4710c9539d5ff","observation_id":"0cba5a80-bd82-4469-ab58-fa7c906b6ecb","resolution":{"observed_at":"2026-08-02T03:56:27.628364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:27.647136Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":118,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.647136Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:d7711b25acacc556365bf1704f8fa917d067bae35093411f8d456bd36a5e7b6a","observation_id":"67dd4cdc-1227-43f2-989e-788c9c1f206f","resolution":{"observed_at":"2026-08-02T03:56:27.647136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-02T03:56:27.724507Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":119,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.724507Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:be15ae1938b72504be31d2c4eadfe2f9759db8cca5749b02ce683e3d8840d92e","observation_id":"7e058d93-dbaa-4d53-9c0c-4d9c58af3a61","resolution":{"observed_at":"2026-08-02T03:56:27.724507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12122","last_updated":"2024-09-18T16:45:37Z","snapshot_observed_at":"2026-07-06T19:17:41.512834Z","submitted_at":"2024-09-18T16:45:37Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12122","snapshot_observed_at":"2026-08-02T03:56:27.820539Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":120,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.820539Z"},"links":{"cited_paper":"/paper/2409.12122","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:31e3dffc9eb955101226169ba2639ee9afe2829886dba31190c0c46eab9a8100","observation_id":"2a416eac-6665-4247-b66c-dc64f2311023","resolution":{"observed_at":"2026-08-02T03:56:27.820539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:27.965607Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.965607Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:8a40aefd31d9fad41963aedeeb1fdec99204fb012b762b1c46a0396d12a095dc","observation_id":"db62ca64-039c-4e18-bb53-303a7ae71964","resolution":{"observed_at":"2026-08-02T03:56:27.965607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:28.073779Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":122,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:28.073779Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:f89555bcaa7470ddbef7e5e004d96839c700000561fd2a833a075d5d176eaa59","observation_id":"1df684f4-74c3-4f48-b436-f3abc4622833","resolution":{"observed_at":"2026-08-02T03:56:28.073779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T10:59:34.790525Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":100,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":108},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 100 of 108 outbound references and 0 inbound Pith citation observations for arXiv:2607.13753."}