{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:A2RVDFE3NVMYT5ELXR5OUOCOHV","short_pith_number":"pith:A2RVDFE3","schema_version":"1.0","canonical_sha256":"06a351949b6d5989f48bbc7aea384e3d6e43459a3c86adf3b5b2bfc0512acb30","source":{"kind":"arxiv","id":"2410.13343","version":1},"attestation_state":"computed","paper":{"title":"Do LLMs Overcome Shortcut Learning? An Evaluation of Shortcut Challenges in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Guangting Zheng, Kai Zhang, Lili Zhao, Qi Liu, Yu Yuan","submitted_at":"2024-10-17T08:52:52Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable capabilities in various natural language processing tasks. However, LLMs may rely on dataset biases as shortcuts for prediction, which can significantly impair their robustness and generalization capabilities. This paper presents Shortcut Suite, a comprehensive test suite designed to evaluate the impact of shortcuts on LLMs' performance, incorporating six shortcut types, five evaluation metrics, and four prompting strategies. Our extensive experiments yield several key findings: 1) LLMs demonstrate varying reliance on shortcuts for downstream "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.13343","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-17T08:52:52Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"775ebfb7f74e5103c7586e82266279a1f250c6b259e1573b568a89721b51104f","abstract_canon_sha256":"ccc30940d762f0467a4a91c66f62042a83d4a29010b408a24f0c6d168e53cb27"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:58.929729Z","signature_b64":"4tOUEhHXYma5mej/BxNb/gFZM6zoFEftNMKTUF9HLI4vyVkanzIClracrVFK6XMOWAM+PhuiQZyvEw3wLucAAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"06a351949b6d5989f48bbc7aea384e3d6e43459a3c86adf3b5b2bfc0512acb30","last_reissued_at":"2026-07-05T09:21:58.929235Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:58.929235Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do LLMs Overcome Shortcut Learning? An Evaluation of Shortcut Challenges in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Guangting Zheng, Kai Zhang, Lili Zhao, Qi Liu, Yu Yuan","submitted_at":"2024-10-17T08:52:52Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable capabilities in various natural language processing tasks. However, LLMs may rely on dataset biases as shortcuts for prediction, which can significantly impair their robustness and generalization capabilities. This paper presents Shortcut Suite, a comprehensive test suite designed to evaluate the impact of shortcuts on LLMs' performance, incorporating six shortcut types, five evaluation metrics, and four prompting strategies. Our extensive experiments yield several key findings: 1) LLMs demonstrate varying reliance on shortcuts for downstream "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.13343","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.13343/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.13343","created_at":"2026-07-05T09:21:58.929301+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.13343v1","created_at":"2026-07-05T09:21:58.929301+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.13343","created_at":"2026-07-05T09:21:58.929301+00:00"},{"alias_kind":"pith_short_12","alias_value":"A2RVDFE3NVMY","created_at":"2026-07-05T09:21:58.929301+00:00"},{"alias_kind":"pith_short_16","alias_value":"A2RVDFE3NVMYT5EL","created_at":"2026-07-05T09:21:58.929301+00:00"},{"alias_kind":"pith_short_8","alias_value":"A2RVDFE3","created_at":"2026-07-05T09:21:58.929301+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28615","citing_title":"What LLMs explain is not what they believe: Evaluating explanation sufficiency under models' own input beliefs","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19774","citing_title":"If Concept Bottlenecks are the Question, are Foundation Models the Answer?","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11482","citing_title":"NeuroFlake: A Neuro-Symbolic LLM Framework for Flaky Test Classification","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02658","citing_title":"Deciphering Shortcut Learning from an Evolutionary Game Theory Perspective","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A2RVDFE3NVMYT5ELXR5OUOCOHV","json":"https://pith.science/pith/A2RVDFE3NVMYT5ELXR5OUOCOHV.json","graph_json":"https://pith.science/api/pith-number/A2RVDFE3NVMYT5ELXR5OUOCOHV/graph.json","events_json":"https://pith.science/api/pith-number/A2RVDFE3NVMYT5ELXR5OUOCOHV/events.json","paper":"https://pith.science/paper/A2RVDFE3"},"agent_actions":{"view_html":"https://pith.science/pith/A2RVDFE3NVMYT5ELXR5OUOCOHV","download_json":"https://pith.science/pith/A2RVDFE3NVMYT5ELXR5OUOCOHV.json","view_paper":"https://pith.science/paper/A2RVDFE3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.13343&json=true","fetch_graph":"https://pith.science/api/pith-number/A2RVDFE3NVMYT5ELXR5OUOCOHV/graph.json","fetch_events":"https://pith.science/api/pith-number/A2RVDFE3NVMYT5ELXR5OUOCOHV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A2RVDFE3NVMYT5ELXR5OUOCOHV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A2RVDFE3NVMYT5ELXR5OUOCOHV/action/storage_attestation","attest_author":"https://pith.science/pith/A2RVDFE3NVMYT5ELXR5OUOCOHV/action/author_attestation","sign_citation":"https://pith.science/pith/A2RVDFE3NVMYT5ELXR5OUOCOHV/action/citation_signature","submit_replication":"https://pith.science/pith/A2RVDFE3NVMYT5ELXR5OUOCOHV/action/replication_record"}},"created_at":"2026-07-05T09:21:58.929301+00:00","updated_at":"2026-07-05T09:21:58.929301+00:00"}