{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YK4EUGHCAOCJT22EEAGXFCCC7E","short_pith_number":"pith:YK4EUGHC","schema_version":"1.0","canonical_sha256":"c2b84a18e2038499eb44200d728842f9342c300c0ae043e40de562cf603578ed","source":{"kind":"arxiv","id":"2307.02477","version":3},"attestation_state":"computed","paper":{"title":"Reasoning or Reciting? Exploring the Capabilities and Limitations of Language Models Through Counterfactual Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alexis Ross, Bailin Wang, Boyuan Chen, Ekin Aky\\\"urek, Jacob Andreas, Linlu Qiu, Najoung Kim, Yoon Kim, Zhaofeng Wu","submitted_at":"2023-07-05T17:50:42Z","abstract_excerpt":"The impressive performance of recent language models across a wide range of tasks suggests that they possess a degree of abstract reasoning skills. Are these skills general and transferable, or specialized to specific tasks seen during pretraining? To disentangle these effects, we propose an evaluation framework based on \"counterfactual\" task variants that deviate from the default assumptions underlying standard tasks. Across a suite of 11 tasks, we observe nontrivial performance on the counterfactual variants, but nevertheless find that performance substantially and consistently degrades comp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.02477","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-07-05T17:50:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"80dd73d9a5eccbbca04c81914a1cca0fda1e90ef6e6a7323e885911ae610d9c4","abstract_canon_sha256":"cd8974d3ab18e555109b0d06d651c279d3835606f6e2bd8d9b9f15922626a77c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:00.022934Z","signature_b64":"NZ0eG3mxBGIAoaSetNEaML0wlOp1ie8JDIDafAGv21fam350mWjdDqENZoYXPw3e+nf0YxH0JJacawE5g6hvAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c2b84a18e2038499eb44200d728842f9342c300c0ae043e40de562cf603578ed","last_reissued_at":"2026-07-05T08:02:00.022396Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:00.022396Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reasoning or Reciting? Exploring the Capabilities and Limitations of Language Models Through Counterfactual Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alexis Ross, Bailin Wang, Boyuan Chen, Ekin Aky\\\"urek, Jacob Andreas, Linlu Qiu, Najoung Kim, Yoon Kim, Zhaofeng Wu","submitted_at":"2023-07-05T17:50:42Z","abstract_excerpt":"The impressive performance of recent language models across a wide range of tasks suggests that they possess a degree of abstract reasoning skills. Are these skills general and transferable, or specialized to specific tasks seen during pretraining? To disentangle these effects, we propose an evaluation framework based on \"counterfactual\" task variants that deviate from the default assumptions underlying standard tasks. Across a suite of 11 tasks, we observe nontrivial performance on the counterfactual variants, but nevertheless find that performance substantially and consistently degrades comp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.02477","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.02477/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.02477","created_at":"2026-07-05T08:02:00.022457+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.02477v3","created_at":"2026-07-05T08:02:00.022457+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.02477","created_at":"2026-07-05T08:02:00.022457+00:00"},{"alias_kind":"pith_short_12","alias_value":"YK4EUGHCAOCJ","created_at":"2026-07-05T08:02:00.022457+00:00"},{"alias_kind":"pith_short_16","alias_value":"YK4EUGHCAOCJT22E","created_at":"2026-07-05T08:02:00.022457+00:00"},{"alias_kind":"pith_short_8","alias_value":"YK4EUGHC","created_at":"2026-07-05T08:02:00.022457+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08068","citing_title":"DICE: Entropy-Regularized Equilibrium Selection for Stable Multi-Agent LLM Coordination","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00276","citing_title":"Testing Frontier Large Language Models' Physics Literacy in Parallel Physical Worlds","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29799","citing_title":"The CRISTAL Method: Neurosymbolic analysis from AI-synthesized world models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02211","citing_title":"Consistency Training while Mitigating Obfuscation via Rate Matching","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2402.09664","citing_title":"CodeMind: Evaluating Large Language Models for Code Reasoning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15079","citing_title":"Assessing Coherency and Consistency of Code Execution Reasoning by Large Language Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07307","citing_title":"Rethinking Dense Sequential Chains: Reasoning Language Models Can Extract Answers from Sparse, Order-Shuffling Chain-of-Thoughts","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13934","citing_title":"Towards Enabling An Artificial Self-Construction Software Life-cycle via Autopoietic Architectures","ref_index":67,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YK4EUGHCAOCJT22EEAGXFCCC7E","json":"https://pith.science/pith/YK4EUGHCAOCJT22EEAGXFCCC7E.json","graph_json":"https://pith.science/api/pith-number/YK4EUGHCAOCJT22EEAGXFCCC7E/graph.json","events_json":"https://pith.science/api/pith-number/YK4EUGHCAOCJT22EEAGXFCCC7E/events.json","paper":"https://pith.science/paper/YK4EUGHC"},"agent_actions":{"view_html":"https://pith.science/pith/YK4EUGHCAOCJT22EEAGXFCCC7E","download_json":"https://pith.science/pith/YK4EUGHCAOCJT22EEAGXFCCC7E.json","view_paper":"https://pith.science/paper/YK4EUGHC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.02477&json=true","fetch_graph":"https://pith.science/api/pith-number/YK4EUGHCAOCJT22EEAGXFCCC7E/graph.json","fetch_events":"https://pith.science/api/pith-number/YK4EUGHCAOCJT22EEAGXFCCC7E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YK4EUGHCAOCJT22EEAGXFCCC7E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YK4EUGHCAOCJT22EEAGXFCCC7E/action/storage_attestation","attest_author":"https://pith.science/pith/YK4EUGHCAOCJT22EEAGXFCCC7E/action/author_attestation","sign_citation":"https://pith.science/pith/YK4EUGHCAOCJT22EEAGXFCCC7E/action/citation_signature","submit_replication":"https://pith.science/pith/YK4EUGHCAOCJT22EEAGXFCCC7E/action/replication_record"}},"created_at":"2026-07-05T08:02:00.022457+00:00","updated_at":"2026-07-05T08:02:00.022457+00:00"}