{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BBD6VCUVETODPC4TIAU6K656AG","short_pith_number":"pith:BBD6VCUV","schema_version":"1.0","canonical_sha256":"0847ea8a9524dc378b934029e57bbe01be6690621bf9f795c5e770b74d88a8a0","source":{"kind":"arxiv","id":"2408.08651","version":3},"attestation_state":"computed","paper":{"title":"Chain of Thought Still Thinks Fast: APriCoT Helps with Thinking Slow","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Douglas Fisher, Jesse Roberts, Kyle Moore, Thao Pham","submitted_at":"2024-08-16T10:34:50Z","abstract_excerpt":"Language models are known to absorb biases from their training data, leading to predictions driven by statistical regularities rather than semantic relevance. We investigate the impact of these biases on answer choice preferences in the Massive Multi-Task Language Understanding (MMLU) task. Our findings show that these biases are predictive of model preference and mirror human test-taking strategies even when chain of thought (CoT) reasoning is used. To address this issue, we introduce Counterfactual Prompting with Agnostically Primed CoT (APriCoT). We demonstrate that while Counterfactual Pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.08651","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-16T10:34:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0ff49c9fef61951937078f5452f98d1e7853f48b94ef29a55ec00f4025adee92","abstract_canon_sha256":"bf4953dbd0e24019ecd3f335cb521199bd449971355cc57eca98fabdcff9a9c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:52:00.332279Z","signature_b64":"B05xAitAQHeCsPII9MCW7KrUpfYcbta1lAbh/ZGtwXEr5zNX8pqTGBe2NXvvwE3qHtvi4FLEZ/8XkPHl10hyCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0847ea8a9524dc378b934029e57bbe01be6690621bf9f795c5e770b74d88a8a0","last_reissued_at":"2026-07-05T11:52:00.331802Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:52:00.331802Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Chain of Thought Still Thinks Fast: APriCoT Helps with Thinking Slow","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Douglas Fisher, Jesse Roberts, Kyle Moore, Thao Pham","submitted_at":"2024-08-16T10:34:50Z","abstract_excerpt":"Language models are known to absorb biases from their training data, leading to predictions driven by statistical regularities rather than semantic relevance. We investigate the impact of these biases on answer choice preferences in the Massive Multi-Task Language Understanding (MMLU) task. Our findings show that these biases are predictive of model preference and mirror human test-taking strategies even when chain of thought (CoT) reasoning is used. To address this issue, we introduce Counterfactual Prompting with Agnostically Primed CoT (APriCoT). We demonstrate that while Counterfactual Pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.08651","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.08651/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.08651","created_at":"2026-07-05T11:52:00.331860+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.08651v3","created_at":"2026-07-05T11:52:00.331860+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.08651","created_at":"2026-07-05T11:52:00.331860+00:00"},{"alias_kind":"pith_short_12","alias_value":"BBD6VCUVETOD","created_at":"2026-07-05T11:52:00.331860+00:00"},{"alias_kind":"pith_short_16","alias_value":"BBD6VCUVETODPC4T","created_at":"2026-07-05T11:52:00.331860+00:00"},{"alias_kind":"pith_short_8","alias_value":"BBD6VCUV","created_at":"2026-07-05T11:52:00.331860+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.10852","citing_title":"LLMs on Trial: Evaluating Judicial Fairness for Large Language Models","ref_index":2022,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BBD6VCUVETODPC4TIAU6K656AG","json":"https://pith.science/pith/BBD6VCUVETODPC4TIAU6K656AG.json","graph_json":"https://pith.science/api/pith-number/BBD6VCUVETODPC4TIAU6K656AG/graph.json","events_json":"https://pith.science/api/pith-number/BBD6VCUVETODPC4TIAU6K656AG/events.json","paper":"https://pith.science/paper/BBD6VCUV"},"agent_actions":{"view_html":"https://pith.science/pith/BBD6VCUVETODPC4TIAU6K656AG","download_json":"https://pith.science/pith/BBD6VCUVETODPC4TIAU6K656AG.json","view_paper":"https://pith.science/paper/BBD6VCUV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.08651&json=true","fetch_graph":"https://pith.science/api/pith-number/BBD6VCUVETODPC4TIAU6K656AG/graph.json","fetch_events":"https://pith.science/api/pith-number/BBD6VCUVETODPC4TIAU6K656AG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BBD6VCUVETODPC4TIAU6K656AG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BBD6VCUVETODPC4TIAU6K656AG/action/storage_attestation","attest_author":"https://pith.science/pith/BBD6VCUVETODPC4TIAU6K656AG/action/author_attestation","sign_citation":"https://pith.science/pith/BBD6VCUVETODPC4TIAU6K656AG/action/citation_signature","submit_replication":"https://pith.science/pith/BBD6VCUVETODPC4TIAU6K656AG/action/replication_record"}},"created_at":"2026-07-05T11:52:00.331860+00:00","updated_at":"2026-07-05T11:52:00.331860+00:00"}