{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AJJ6F6JXDBVDTRLNYM7PLKV2JE","short_pith_number":"pith:AJJ6F6JX","schema_version":"1.0","canonical_sha256":"0253e2f937186a39c56dc33ef5aaba49223e97c474a13e70fdc2b9ad989937a6","source":{"kind":"arxiv","id":"2403.05518","version":3},"attestation_state":"computed","paper":{"title":"Bias-Augmented Consistency Training Reduces Biased Reasoning in Chain-of-Thought","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Edward Rees, Ethan Perez, Hunar Batra, James Chua, Julian Michael, Miles Turpin, Samuel R. Bowman","submitted_at":"2024-03-08T18:41:42Z","abstract_excerpt":"Chain-of-thought prompting (CoT) has the potential to improve the explainability of language model reasoning. But CoT can also systematically misrepresent the factors influencing models' behavior -- for example, rationalizing answers in line with a user's opinion.\n  We first create a new dataset of 9 different biases that affect GPT-3.5-Turbo and Llama-8b models. These consist of spurious-few-shot patterns, post hoc rationalization, and sycophantic settings. Models switch to the answer implied by the bias, without mentioning the effect of the bias in the CoT.\n  To mitigate this biased reasonin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.05518","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-08T18:41:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f294919a2003da12962afae7c1b397941726ec7e7887feb3748e6bf1ac2658d6","abstract_canon_sha256":"fcf04ad9c5125a486136ff842b2f245cc2349a6d2e950151bd61b7aa2446a445"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:48.789767Z","signature_b64":"BBt8ISUnX8ZZVwavEqWH2mYpFW+YTqfWaLN1KbdVOHmGdOpYBF2O1mszHZH5uJT+NG9hJYeBEt16V9cEuwUPBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0253e2f937186a39c56dc33ef5aaba49223e97c474a13e70fdc2b9ad989937a6","last_reissued_at":"2026-07-05T11:27:48.789266Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:48.789266Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bias-Augmented Consistency Training Reduces Biased Reasoning in Chain-of-Thought","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Edward Rees, Ethan Perez, Hunar Batra, James Chua, Julian Michael, Miles Turpin, Samuel R. Bowman","submitted_at":"2024-03-08T18:41:42Z","abstract_excerpt":"Chain-of-thought prompting (CoT) has the potential to improve the explainability of language model reasoning. But CoT can also systematically misrepresent the factors influencing models' behavior -- for example, rationalizing answers in line with a user's opinion.\n  We first create a new dataset of 9 different biases that affect GPT-3.5-Turbo and Llama-8b models. These consist of spurious-few-shot patterns, post hoc rationalization, and sycophantic settings. Models switch to the answer implied by the bias, without mentioning the effect of the bias in the CoT.\n  To mitigate this biased reasonin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.05518","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.05518/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.05518","created_at":"2026-07-05T11:27:48.789342+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.05518v3","created_at":"2026-07-05T11:27:48.789342+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.05518","created_at":"2026-07-05T11:27:48.789342+00:00"},{"alias_kind":"pith_short_12","alias_value":"AJJ6F6JXDBVD","created_at":"2026-07-05T11:27:48.789342+00:00"},{"alias_kind":"pith_short_16","alias_value":"AJJ6F6JXDBVDTRLN","created_at":"2026-07-05T11:27:48.789342+00:00"},{"alias_kind":"pith_short_8","alias_value":"AJJ6F6JX","created_at":"2026-07-05T11:27:48.789342+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03810","citing_title":"Consistency Training Can Entrench Misalignment","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03226","citing_title":"Self-Mined Hardness for Safety Fine-Tuning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24279","citing_title":"ContextEcho: A Benchmark for Persona Drift in Long Agentic-Coding Sessions","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28467","citing_title":"Mitigating Adaptive Attacks against Reasoning Models with Activation Consistency Training","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21834","citing_title":"On-Policy Consistency Training Improves LLM Safety with Minimal Capability Degradation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03226","citing_title":"Self-Mined Hardness for Safety Fine-Tuning","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AJJ6F6JXDBVDTRLNYM7PLKV2JE","json":"https://pith.science/pith/AJJ6F6JXDBVDTRLNYM7PLKV2JE.json","graph_json":"https://pith.science/api/pith-number/AJJ6F6JXDBVDTRLNYM7PLKV2JE/graph.json","events_json":"https://pith.science/api/pith-number/AJJ6F6JXDBVDTRLNYM7PLKV2JE/events.json","paper":"https://pith.science/paper/AJJ6F6JX"},"agent_actions":{"view_html":"https://pith.science/pith/AJJ6F6JXDBVDTRLNYM7PLKV2JE","download_json":"https://pith.science/pith/AJJ6F6JXDBVDTRLNYM7PLKV2JE.json","view_paper":"https://pith.science/paper/AJJ6F6JX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.05518&json=true","fetch_graph":"https://pith.science/api/pith-number/AJJ6F6JXDBVDTRLNYM7PLKV2JE/graph.json","fetch_events":"https://pith.science/api/pith-number/AJJ6F6JXDBVDTRLNYM7PLKV2JE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AJJ6F6JXDBVDTRLNYM7PLKV2JE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AJJ6F6JXDBVDTRLNYM7PLKV2JE/action/storage_attestation","attest_author":"https://pith.science/pith/AJJ6F6JXDBVDTRLNYM7PLKV2JE/action/author_attestation","sign_citation":"https://pith.science/pith/AJJ6F6JXDBVDTRLNYM7PLKV2JE/action/citation_signature","submit_replication":"https://pith.science/pith/AJJ6F6JXDBVDTRLNYM7PLKV2JE/action/replication_record"}},"created_at":"2026-07-05T11:27:48.789342+00:00","updated_at":"2026-07-05T11:27:48.789342+00:00"}