{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AAOOEENJNSUXLNO4CB7XAOSG32","short_pith_number":"pith:AAOOEENJ","schema_version":"1.0","canonical_sha256":"001ce211a96ca975b5dc107f703a46de9cb58fc5091ea143333d61e0c68a8683","source":{"kind":"arxiv","id":"2402.13950","version":4},"attestation_state":"computed","paper":{"title":"Making Reasoning Matter: Measuring and Improving Faithfulness of Chain-of-Thought Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Antoine Bosselut, Boi Faltings, Debjit Paul, Robert West","submitted_at":"2024-02-21T17:23:59Z","abstract_excerpt":"Large language models (LLMs) have been shown to perform better when asked to reason step-by-step before answering a question. However, it is unclear to what degree the model's final answer is faithful to the stated reasoning steps. In this paper, we perform a causal mediation analysis on twelve LLMs to examine how intermediate reasoning steps generated by the LLM influence the final outcome and find that LLMs do not reliably use their intermediate reasoning steps when generating an answer. To address this issue, we introduce FRODO, a framework to tailor small-sized LMs to generate correct reas"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.13950","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-21T17:23:59Z","cross_cats_sorted":[],"title_canon_sha256":"a1b9cb7cedf82ac6bdaffd259851c47c54b6e2f61865d183e1b7d7cb6a6c5516","abstract_canon_sha256":"f2d40bd80a130a1c7257a5ccb40f267651dbeb58371118ed8efcb34877096189"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:13.016712Z","signature_b64":"fb36Mg9gIjT4+bL6QMfG6ItNiYbLGUGw5iUW+s/hn9qvWCuvtGstkiTO/khm3iHerUrwye/Sk7vOtlBwonxaCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"001ce211a96ca975b5dc107f703a46de9cb58fc5091ea143333d61e0c68a8683","last_reissued_at":"2026-07-05T09:16:13.016219Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:13.016219Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Making Reasoning Matter: Measuring and Improving Faithfulness of Chain-of-Thought Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Antoine Bosselut, Boi Faltings, Debjit Paul, Robert West","submitted_at":"2024-02-21T17:23:59Z","abstract_excerpt":"Large language models (LLMs) have been shown to perform better when asked to reason step-by-step before answering a question. However, it is unclear to what degree the model's final answer is faithful to the stated reasoning steps. In this paper, we perform a causal mediation analysis on twelve LLMs to examine how intermediate reasoning steps generated by the LLM influence the final outcome and find that LLMs do not reliably use their intermediate reasoning steps when generating an answer. To address this issue, we introduce FRODO, a framework to tailor small-sized LMs to generate correct reas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.13950","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.13950/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.13950","created_at":"2026-07-05T09:16:13.016274+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.13950v4","created_at":"2026-07-05T09:16:13.016274+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.13950","created_at":"2026-07-05T09:16:13.016274+00:00"},{"alias_kind":"pith_short_12","alias_value":"AAOOEENJNSUX","created_at":"2026-07-05T09:16:13.016274+00:00"},{"alias_kind":"pith_short_16","alias_value":"AAOOEENJNSUXLNO4","created_at":"2026-07-05T09:16:13.016274+00:00"},{"alias_kind":"pith_short_8","alias_value":"AAOOEENJ","created_at":"2026-07-05T09:16:13.016274+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24661","citing_title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00924","citing_title":"Graph-Native Reinforcement Learning Enables Traceable Scientific Hypothesis Generation through Conceptual Recombination","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24661","citing_title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24661","citing_title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25603","citing_title":"Detecting Unfaithful Chain-of-Thought via Circuit-Guided Internal-External Discrepancy","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13392","citing_title":"ReSS: Learning Reasoning Models for Tabular Data Prediction via Symbolic Scaffold","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12824","citing_title":"Mechanism Plausibility in Generative Agent-Based Modeling","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07407","citing_title":"Training Language Models to Use Prolog as a Tool","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12824","citing_title":"Mechanism Plausibility in Generative Agent-Based Modeling","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01704","citing_title":"The Reasoning Trap: An Information-Theoretic Bound on Closed-System Multi-Step LLM Reasoning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13392","citing_title":"ReSS: Learning Reasoning Models for Tabular Data Prediction via Symbolic Scaffold","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AAOOEENJNSUXLNO4CB7XAOSG32","json":"https://pith.science/pith/AAOOEENJNSUXLNO4CB7XAOSG32.json","graph_json":"https://pith.science/api/pith-number/AAOOEENJNSUXLNO4CB7XAOSG32/graph.json","events_json":"https://pith.science/api/pith-number/AAOOEENJNSUXLNO4CB7XAOSG32/events.json","paper":"https://pith.science/paper/AAOOEENJ"},"agent_actions":{"view_html":"https://pith.science/pith/AAOOEENJNSUXLNO4CB7XAOSG32","download_json":"https://pith.science/pith/AAOOEENJNSUXLNO4CB7XAOSG32.json","view_paper":"https://pith.science/paper/AAOOEENJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.13950&json=true","fetch_graph":"https://pith.science/api/pith-number/AAOOEENJNSUXLNO4CB7XAOSG32/graph.json","fetch_events":"https://pith.science/api/pith-number/AAOOEENJNSUXLNO4CB7XAOSG32/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AAOOEENJNSUXLNO4CB7XAOSG32/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AAOOEENJNSUXLNO4CB7XAOSG32/action/storage_attestation","attest_author":"https://pith.science/pith/AAOOEENJNSUXLNO4CB7XAOSG32/action/author_attestation","sign_citation":"https://pith.science/pith/AAOOEENJNSUXLNO4CB7XAOSG32/action/citation_signature","submit_replication":"https://pith.science/pith/AAOOEENJNSUXLNO4CB7XAOSG32/action/replication_record"}},"created_at":"2026-07-05T09:16:13.016274+00:00","updated_at":"2026-07-05T09:16:13.016274+00:00"}