{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XNCSOJBCTBXOSEG7MAHWLZMALL","short_pith_number":"pith:XNCSOJBC","schema_version":"1.0","canonical_sha256":"bb45272422986ee910df600f65e5805af9a1047e2f80b4668514fc64b64787b1","source":{"kind":"arxiv","id":"2411.12580","version":2},"attestation_state":"computed","paper":{"title":"Procedural Knowledge in Pretraining Drives Reasoning in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Acyr Locatelli, Dwarak Talupuru, Edward Grefenstette, Juhan Bae, Laura Ruis, Max Bartolo, Maximilian Mozes, Robert Kirk, Siddhartha Rao Kamalakara, Tim Rockt\\\"aschel","submitted_at":"2024-11-19T15:47:12Z","abstract_excerpt":"The capabilities and limitations of Large Language Models have been sketched out in great detail in recent years, providing an intriguing yet conflicting picture. On the one hand, LLMs demonstrate a general ability to solve problems. On the other hand, they show surprising reasoning gaps when compared to humans, casting doubt on the robustness of their generalisation strategies. The sheer volume of data used in the design of LLMs has precluded us from applying the method traditionally used to measure generalisation: train-test set separation. To overcome this, we study what kind of generalisat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.12580","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-19T15:47:12Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1a0711b5a7aa216e4e339307d143e20dc5719d7e2d4887825e9f2540c95e0ec9","abstract_canon_sha256":"05426405b7d5279e8a65cd35141df8c39b0670be823b4c41948d4f693e398374"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:14.036862Z","signature_b64":"McozKep61gmQC9jOZf/fj0Je+M0VC2+LIGdPr2r1F6gZIfKePsJpENUWKLRYrpcycpdF/n4IQ30RJOLZtFIlDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb45272422986ee910df600f65e5805af9a1047e2f80b4668514fc64b64787b1","last_reissued_at":"2026-07-05T10:25:14.036198Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:14.036198Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Procedural Knowledge in Pretraining Drives Reasoning in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Acyr Locatelli, Dwarak Talupuru, Edward Grefenstette, Juhan Bae, Laura Ruis, Max Bartolo, Maximilian Mozes, Robert Kirk, Siddhartha Rao Kamalakara, Tim Rockt\\\"aschel","submitted_at":"2024-11-19T15:47:12Z","abstract_excerpt":"The capabilities and limitations of Large Language Models have been sketched out in great detail in recent years, providing an intriguing yet conflicting picture. On the one hand, LLMs demonstrate a general ability to solve problems. On the other hand, they show surprising reasoning gaps when compared to humans, casting doubt on the robustness of their generalisation strategies. The sheer volume of data used in the design of LLMs has precluded us from applying the method traditionally used to measure generalisation: train-test set separation. To overcome this, we study what kind of generalisat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.12580","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.12580/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.12580","created_at":"2026-07-05T10:25:14.036269+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.12580v2","created_at":"2026-07-05T10:25:14.036269+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.12580","created_at":"2026-07-05T10:25:14.036269+00:00"},{"alias_kind":"pith_short_12","alias_value":"XNCSOJBCTBXO","created_at":"2026-07-05T10:25:14.036269+00:00"},{"alias_kind":"pith_short_16","alias_value":"XNCSOJBCTBXOSEG7","created_at":"2026-07-05T10:25:14.036269+00:00"},{"alias_kind":"pith_short_8","alias_value":"XNCSOJBC","created_at":"2026-07-05T10:25:14.036269+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23591","citing_title":"Quantifying the Agreement Between Data-Influence and Data-Similarity to Understand LLM Behavior","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2602.24176","citing_title":"Beyond Explainable AI (XAI): An Overdue Paradigm Shift and Post-XAI Research Directions","ref_index":148,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XNCSOJBCTBXOSEG7MAHWLZMALL","json":"https://pith.science/pith/XNCSOJBCTBXOSEG7MAHWLZMALL.json","graph_json":"https://pith.science/api/pith-number/XNCSOJBCTBXOSEG7MAHWLZMALL/graph.json","events_json":"https://pith.science/api/pith-number/XNCSOJBCTBXOSEG7MAHWLZMALL/events.json","paper":"https://pith.science/paper/XNCSOJBC"},"agent_actions":{"view_html":"https://pith.science/pith/XNCSOJBCTBXOSEG7MAHWLZMALL","download_json":"https://pith.science/pith/XNCSOJBCTBXOSEG7MAHWLZMALL.json","view_paper":"https://pith.science/paper/XNCSOJBC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.12580&json=true","fetch_graph":"https://pith.science/api/pith-number/XNCSOJBCTBXOSEG7MAHWLZMALL/graph.json","fetch_events":"https://pith.science/api/pith-number/XNCSOJBCTBXOSEG7MAHWLZMALL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XNCSOJBCTBXOSEG7MAHWLZMALL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XNCSOJBCTBXOSEG7MAHWLZMALL/action/storage_attestation","attest_author":"https://pith.science/pith/XNCSOJBCTBXOSEG7MAHWLZMALL/action/author_attestation","sign_citation":"https://pith.science/pith/XNCSOJBCTBXOSEG7MAHWLZMALL/action/citation_signature","submit_replication":"https://pith.science/pith/XNCSOJBCTBXOSEG7MAHWLZMALL/action/replication_record"}},"created_at":"2026-07-05T10:25:14.036269+00:00","updated_at":"2026-07-05T10:25:14.036269+00:00"}