{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:W24FZY5BPDSL2OM5GEX6OJPBQD","short_pith_number":"pith:W24FZY5B","schema_version":"1.0","canonical_sha256":"b6b85ce3a178e4bd399d312fe725e180cfc2f8b8ef9b1281ad57796e55adf84c","source":{"kind":"arxiv","id":"2504.05518","version":1},"attestation_state":"computed","paper":{"title":"Evaluating the Generalization Capabilities of Large Language Models on Code Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Julian Dai, Martin Rinard, Nikos Vasilakis, Rem Yang","submitted_at":"2025-04-07T21:25:31Z","abstract_excerpt":"We assess how the code reasoning abilities of large language models (LLMs) generalize to different kinds of programs. We present techniques for obtaining in- and out-of-distribution programs with different characteristics: code sampled from a domain-specific language, code automatically generated by an LLM, code collected from competitive programming contests, and mutated versions of these programs. We also present an experimental methodology for evaluating LLM generalization by comparing their performance on these programs. We perform an extensive evaluation across 10 state-of-the-art models "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.05518","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-04-07T21:25:31Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"845a285a4e39136c5e3dcf1aa62327d159b4003b46d23cd288b0821f97c14807","abstract_canon_sha256":"ad7781cc9a3ed20f95fa3621ecc998d91be3b635232c78fe8f5fbe6645419d36"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:46:05.106374Z","signature_b64":"YSax7L0USZHUAg4ia/OTagglm8z6lZcvctcRT6gFEDic/1Tkn50L2qc9uS7Pr6x2/YGrVNMibrfFszve9oqEDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6b85ce3a178e4bd399d312fe725e180cfc2f8b8ef9b1281ad57796e55adf84c","last_reissued_at":"2026-07-05T10:46:05.105895Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:46:05.105895Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating the Generalization Capabilities of Large Language Models on Code Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Julian Dai, Martin Rinard, Nikos Vasilakis, Rem Yang","submitted_at":"2025-04-07T21:25:31Z","abstract_excerpt":"We assess how the code reasoning abilities of large language models (LLMs) generalize to different kinds of programs. We present techniques for obtaining in- and out-of-distribution programs with different characteristics: code sampled from a domain-specific language, code automatically generated by an LLM, code collected from competitive programming contests, and mutated versions of these programs. We also present an experimental methodology for evaluating LLM generalization by comparing their performance on these programs. We perform an extensive evaluation across 10 state-of-the-art models "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.05518","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.05518/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.05518","created_at":"2026-07-05T10:46:05.105952+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.05518v1","created_at":"2026-07-05T10:46:05.105952+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.05518","created_at":"2026-07-05T10:46:05.105952+00:00"},{"alias_kind":"pith_short_12","alias_value":"W24FZY5BPDSL","created_at":"2026-07-05T10:46:05.105952+00:00"},{"alias_kind":"pith_short_16","alias_value":"W24FZY5BPDSL2OM5","created_at":"2026-07-05T10:46:05.105952+00:00"},{"alias_kind":"pith_short_8","alias_value":"W24FZY5B","created_at":"2026-07-05T10:46:05.105952+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.14917","citing_title":"Evaluating Code Reasoning Abilities of Large Language Models Under Real-World Settings","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20811","citing_title":"Diagnosing CFG Interpretation in LLMs","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W24FZY5BPDSL2OM5GEX6OJPBQD","json":"https://pith.science/pith/W24FZY5BPDSL2OM5GEX6OJPBQD.json","graph_json":"https://pith.science/api/pith-number/W24FZY5BPDSL2OM5GEX6OJPBQD/graph.json","events_json":"https://pith.science/api/pith-number/W24FZY5BPDSL2OM5GEX6OJPBQD/events.json","paper":"https://pith.science/paper/W24FZY5B"},"agent_actions":{"view_html":"https://pith.science/pith/W24FZY5BPDSL2OM5GEX6OJPBQD","download_json":"https://pith.science/pith/W24FZY5BPDSL2OM5GEX6OJPBQD.json","view_paper":"https://pith.science/paper/W24FZY5B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.05518&json=true","fetch_graph":"https://pith.science/api/pith-number/W24FZY5BPDSL2OM5GEX6OJPBQD/graph.json","fetch_events":"https://pith.science/api/pith-number/W24FZY5BPDSL2OM5GEX6OJPBQD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W24FZY5BPDSL2OM5GEX6OJPBQD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W24FZY5BPDSL2OM5GEX6OJPBQD/action/storage_attestation","attest_author":"https://pith.science/pith/W24FZY5BPDSL2OM5GEX6OJPBQD/action/author_attestation","sign_citation":"https://pith.science/pith/W24FZY5BPDSL2OM5GEX6OJPBQD/action/citation_signature","submit_replication":"https://pith.science/pith/W24FZY5BPDSL2OM5GEX6OJPBQD/action/replication_record"}},"created_at":"2026-07-05T10:46:05.105952+00:00","updated_at":"2026-07-05T10:46:05.105952+00:00"}