{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:U2VLUP4WYX75TZCLO4AYLQAYEU","short_pith_number":"pith:U2VLUP4W","schema_version":"1.0","canonical_sha256":"a6aaba3f96c5ffd9e44b770185c018252b88f1ec52d43fbc533e960b6dcb0cfa","source":{"kind":"arxiv","id":"2501.08156","version":5},"attestation_state":"computed","paper":{"title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"James Chua, Owain Evans","submitted_at":"2025-01-14T14:31:45Z","abstract_excerpt":"Language models trained to solve reasoning tasks via reinforcement learning have achieved striking results. We refer to these models as reasoning models. Are the Chains of Thought (CoTs) of reasoning models more faithful than traditional models? We evaluate three reasoning models (based on Qwen-2.5, Gemini-2, and DeepSeek-V3-Base) on an existing test of faithful CoT. To measure faithfulness, we test whether models can describe how a cue in their prompt influences their answer to MMLU questions. For example, when the cue \"A Stanford Professor thinks the answer is D\" is added to the prompt, mode"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.08156","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-14T14:31:45Z","cross_cats_sorted":[],"title_canon_sha256":"244e40bc05c14c37035f8bbb6e6019a431a70a47c6a25885067c15048cc1f379","abstract_canon_sha256":"ed9546aaaa47993e1b5560442951fd60ae7332aafff6d2980df2946202529c82"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:37:14.906739Z","signature_b64":"JDbI4rg3U76s+IWAfCgrFpOIcqoYFxU/zxp2SbpKQ5A7qmD1h7ErZyuCYX9tTwlsEoOOvFO98IogNYVppLdRDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6aaba3f96c5ffd9e44b770185c018252b88f1ec52d43fbc533e960b6dcb0cfa","last_reissued_at":"2026-07-05T11:37:14.906124Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:37:14.906124Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"James Chua, Owain Evans","submitted_at":"2025-01-14T14:31:45Z","abstract_excerpt":"Language models trained to solve reasoning tasks via reinforcement learning have achieved striking results. We refer to these models as reasoning models. Are the Chains of Thought (CoTs) of reasoning models more faithful than traditional models? We evaluate three reasoning models (based on Qwen-2.5, Gemini-2, and DeepSeek-V3-Base) on an existing test of faithful CoT. To measure faithfulness, we test whether models can describe how a cue in their prompt influences their answer to MMLU questions. For example, when the cue \"A Stanford Professor thinks the answer is D\" is added to the prompt, mode"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.08156","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.08156/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.08156","created_at":"2026-07-05T11:37:14.906197+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.08156v5","created_at":"2026-07-05T11:37:14.906197+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.08156","created_at":"2026-07-05T11:37:14.906197+00:00"},{"alias_kind":"pith_short_12","alias_value":"U2VLUP4WYX75","created_at":"2026-07-05T11:37:14.906197+00:00"},{"alias_kind":"pith_short_16","alias_value":"U2VLUP4WYX75TZCL","created_at":"2026-07-05T11:37:14.906197+00:00"},{"alias_kind":"pith_short_8","alias_value":"U2VLUP4W","created_at":"2026-07-05T11:37:14.906197+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08173","citing_title":"Overthinking: Amplifying Reasoning Weights to Extract Learned Secrets","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2605.24286","citing_title":"Faithfulness as Information Flow: Evaluating and Training Faithful Chain-of-Thought Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24960","citing_title":"Investigating the Interplay between Contextual and Parametric Chain-of-Thought Faithfulness under Optimization","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27251","citing_title":"Compliance versus Sensibility: On the Reasoning Controllability in Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15726","citing_title":"LLM Reasoning Is Latent, Not the Chain of Thought","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U2VLUP4WYX75TZCLO4AYLQAYEU","json":"https://pith.science/pith/U2VLUP4WYX75TZCLO4AYLQAYEU.json","graph_json":"https://pith.science/api/pith-number/U2VLUP4WYX75TZCLO4AYLQAYEU/graph.json","events_json":"https://pith.science/api/pith-number/U2VLUP4WYX75TZCLO4AYLQAYEU/events.json","paper":"https://pith.science/paper/U2VLUP4W"},"agent_actions":{"view_html":"https://pith.science/pith/U2VLUP4WYX75TZCLO4AYLQAYEU","download_json":"https://pith.science/pith/U2VLUP4WYX75TZCLO4AYLQAYEU.json","view_paper":"https://pith.science/paper/U2VLUP4W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.08156&json=true","fetch_graph":"https://pith.science/api/pith-number/U2VLUP4WYX75TZCLO4AYLQAYEU/graph.json","fetch_events":"https://pith.science/api/pith-number/U2VLUP4WYX75TZCLO4AYLQAYEU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U2VLUP4WYX75TZCLO4AYLQAYEU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U2VLUP4WYX75TZCLO4AYLQAYEU/action/storage_attestation","attest_author":"https://pith.science/pith/U2VLUP4WYX75TZCLO4AYLQAYEU/action/author_attestation","sign_citation":"https://pith.science/pith/U2VLUP4WYX75TZCLO4AYLQAYEU/action/citation_signature","submit_replication":"https://pith.science/pith/U2VLUP4WYX75TZCLO4AYLQAYEU/action/replication_record"}},"created_at":"2026-07-05T11:37:14.906197+00:00","updated_at":"2026-07-05T11:37:14.906197+00:00"}