{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4X5MS6QVZ6NKJS7KPUI74ARLJB","short_pith_number":"pith:4X5MS6QV","schema_version":"1.0","canonical_sha256":"e5fac97a15cf9aa4cbea7d11fe022b4869c566673e4cbf15f8739cdd0db7103c","source":{"kind":"arxiv","id":"2405.15092","version":2},"attestation_state":"computed","paper":{"title":"Dissociation of Faithful and Unfaithful Reasoning in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Alice Li, Chenyu Tang, Evelyn Yee, Leon Bergen, Ramamohan Paturi, Yeon Ho Jung","submitted_at":"2024-05-23T22:38:58Z","abstract_excerpt":"Large language models (LLMs) often improve their performance in downstream tasks when they generate Chain of Thought reasoning text before producing an answer. We investigate how LLMs recover from errors in Chain of Thought. Through analysis of error recovery behaviors, we find evidence for unfaithfulness in Chain of Thought, which occurs when models arrive at the correct answer despite invalid reasoning text. We identify factors that shift LLM recovery behavior: LLMs recover more frequently from obvious errors and in contexts that provide more evidence for the correct answer. Critically, thes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.15092","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-05-23T22:38:58Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"19780239cb52bb91c1024c8517611d5b2bbde6e4bb170fdd85e99bad8a7cfd0a","abstract_canon_sha256":"01da5f2a781b2b93532ffc989dcc43e0cf57075ab10886b31400a375b2091664"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:02:19.841012Z","signature_b64":"oV+D9lw3bn8O3nCTsNKkcwmZc9Lfo/xIm66F7EDXuru462pyzsCuEyn8EHTUsUT4OrEVGajnCDOWvcD8JPspCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e5fac97a15cf9aa4cbea7d11fe022b4869c566673e4cbf15f8739cdd0db7103c","last_reissued_at":"2026-07-05T09:02:19.840557Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:02:19.840557Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dissociation of Faithful and Unfaithful Reasoning in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Alice Li, Chenyu Tang, Evelyn Yee, Leon Bergen, Ramamohan Paturi, Yeon Ho Jung","submitted_at":"2024-05-23T22:38:58Z","abstract_excerpt":"Large language models (LLMs) often improve their performance in downstream tasks when they generate Chain of Thought reasoning text before producing an answer. We investigate how LLMs recover from errors in Chain of Thought. Through analysis of error recovery behaviors, we find evidence for unfaithfulness in Chain of Thought, which occurs when models arrive at the correct answer despite invalid reasoning text. We identify factors that shift LLM recovery behavior: LLMs recover more frequently from obvious errors and in contexts that provide more evidence for the correct answer. Critically, thes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.15092","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.15092/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.15092","created_at":"2026-07-05T09:02:19.840613+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.15092v2","created_at":"2026-07-05T09:02:19.840613+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.15092","created_at":"2026-07-05T09:02:19.840613+00:00"},{"alias_kind":"pith_short_12","alias_value":"4X5MS6QVZ6NK","created_at":"2026-07-05T09:02:19.840613+00:00"},{"alias_kind":"pith_short_16","alias_value":"4X5MS6QVZ6NKJS7K","created_at":"2026-07-05T09:02:19.840613+00:00"},{"alias_kind":"pith_short_8","alias_value":"4X5MS6QV","created_at":"2026-07-05T09:02:19.840613+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11445","citing_title":"Forecasting Future Behavior as a Learning Task","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05402","citing_title":"ReasoningFlow: Discourse Structures for Understanding LLM Reasoning Traces","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25603","citing_title":"Detecting Unfaithful Chain-of-Thought via Circuit-Guided Internal-External Discrepancy","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23414","citing_title":"When Planning Fails Despite Correct Execution: On Epistemic Calibration for LLM-Based Multi-Agent Systems","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2510.24941","citing_title":"Can Aha Moments Be Fake? Towards Quantifying Decorative and True Thinking in Chain-of-Thought","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11388","citing_title":"Deep Reasoning in General Purpose Agents via Structured Meta-Cognition","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02819","citing_title":"SCPRM: A Schema-aware Cumulative Process Reward Model for Knowledge Graph Question Answering","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4X5MS6QVZ6NKJS7KPUI74ARLJB","json":"https://pith.science/pith/4X5MS6QVZ6NKJS7KPUI74ARLJB.json","graph_json":"https://pith.science/api/pith-number/4X5MS6QVZ6NKJS7KPUI74ARLJB/graph.json","events_json":"https://pith.science/api/pith-number/4X5MS6QVZ6NKJS7KPUI74ARLJB/events.json","paper":"https://pith.science/paper/4X5MS6QV"},"agent_actions":{"view_html":"https://pith.science/pith/4X5MS6QVZ6NKJS7KPUI74ARLJB","download_json":"https://pith.science/pith/4X5MS6QVZ6NKJS7KPUI74ARLJB.json","view_paper":"https://pith.science/paper/4X5MS6QV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.15092&json=true","fetch_graph":"https://pith.science/api/pith-number/4X5MS6QVZ6NKJS7KPUI74ARLJB/graph.json","fetch_events":"https://pith.science/api/pith-number/4X5MS6QVZ6NKJS7KPUI74ARLJB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4X5MS6QVZ6NKJS7KPUI74ARLJB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4X5MS6QVZ6NKJS7KPUI74ARLJB/action/storage_attestation","attest_author":"https://pith.science/pith/4X5MS6QVZ6NKJS7KPUI74ARLJB/action/author_attestation","sign_citation":"https://pith.science/pith/4X5MS6QVZ6NKJS7KPUI74ARLJB/action/citation_signature","submit_replication":"https://pith.science/pith/4X5MS6QVZ6NKJS7KPUI74ARLJB/action/replication_record"}},"created_at":"2026-07-05T09:02:19.840613+00:00","updated_at":"2026-07-05T09:02:19.840613+00:00"}