{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BEWL46TUZFEZK5FQB5EXDQFMAP","short_pith_number":"pith:BEWL46TU","schema_version":"1.0","canonical_sha256":"092cbe7a74c9499574b00f4971c0ac03cc029c594e4a678dc333f91eb7815eb9","source":{"kind":"arxiv","id":"2404.14662","version":1},"attestation_state":"computed","paper":{"title":"NExT: Teaching Large Language Models to Reason about Code Execution","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.PL","cs.SE"],"primary_cat":"cs.LG","authors_text":"Ansong Ni, Arman Cohan, Charles Sutton, Kensen Shi, Miltiadis Allamanis, Pengcheng Yin, Yinlin Deng","submitted_at":"2024-04-23T01:46:32Z","abstract_excerpt":"A fundamental skill among human developers is the ability to understand and reason about program execution. As an example, a programmer can mentally simulate code execution in natural language to debug and repair code (aka. rubber duck debugging). However, large language models (LLMs) of code are typically trained on the surface textual form of programs, thus may lack a semantic understanding of how programs execute at run-time. To address this issue, we propose NExT, a method to teach LLMs to inspect the execution traces of programs (variable states of executed lines) and reason about their r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.14662","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-04-23T01:46:32Z","cross_cats_sorted":["cs.CL","cs.PL","cs.SE"],"title_canon_sha256":"df33ff89f58c7eb33230f365b1c7b0ed47a3f524194a8758f8375af640caef81","abstract_canon_sha256":"85f28dcb0c6e41a1065217f57bd2fdc4a5555b1ba6a7a28e0b5a5ae27ffa8f5b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:11:08.291523Z","signature_b64":"eVm50GqW3twgYnHsS8kp3nB6glHj9lPTjlceyLhAVP4zdeIeBvs353odFOT9ol27cSTfzs1CLjcb08zjzibkAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"092cbe7a74c9499574b00f4971c0ac03cc029c594e4a678dc333f91eb7815eb9","last_reissued_at":"2026-07-05T08:11:08.291107Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:11:08.291107Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NExT: Teaching Large Language Models to Reason about Code Execution","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.PL","cs.SE"],"primary_cat":"cs.LG","authors_text":"Ansong Ni, Arman Cohan, Charles Sutton, Kensen Shi, Miltiadis Allamanis, Pengcheng Yin, Yinlin Deng","submitted_at":"2024-04-23T01:46:32Z","abstract_excerpt":"A fundamental skill among human developers is the ability to understand and reason about program execution. As an example, a programmer can mentally simulate code execution in natural language to debug and repair code (aka. rubber duck debugging). However, large language models (LLMs) of code are typically trained on the surface textual form of programs, thus may lack a semantic understanding of how programs execute at run-time. To address this issue, we propose NExT, a method to teach LLMs to inspect the execution traces of programs (variable states of executed lines) and reason about their r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.14662","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.14662/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.14662","created_at":"2026-07-05T08:11:08.291166+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.14662v1","created_at":"2026-07-05T08:11:08.291166+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.14662","created_at":"2026-07-05T08:11:08.291166+00:00"},{"alias_kind":"pith_short_12","alias_value":"BEWL46TUZFEZ","created_at":"2026-07-05T08:11:08.291166+00:00"},{"alias_kind":"pith_short_16","alias_value":"BEWL46TUZFEZK5FQ","created_at":"2026-07-05T08:11:08.291166+00:00"},{"alias_kind":"pith_short_8","alias_value":"BEWL46TU","created_at":"2026-07-05T08:11:08.291166+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29742","citing_title":"MicroAgent: Context-Augmented Multi-Agent Framework for Automatic Microservice Decomposition","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2402.09664","citing_title":"CodeMind: Evaluating Large Language Models for Code Reasoning","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18747","citing_title":"Code as Agent Harness","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17444","citing_title":"MemRepair: Hierarchical Memory for Agentic Repository-Level Vulnerability Repair","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2409.12917","citing_title":"Training Language Models to Self-Correct via Reinforcement Learning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11922","citing_title":"StepCodeReasoner: Aligning Code Reasoning with Stepwise Execution Traces via Reinforcement Learning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09186","citing_title":"Agentic MIP Research: Accelerated Constraint Handler Generation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06184","citing_title":"Teaching LLMs Program Semantics via Symbolic Execution Traces","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BEWL46TUZFEZK5FQB5EXDQFMAP","json":"https://pith.science/pith/BEWL46TUZFEZK5FQB5EXDQFMAP.json","graph_json":"https://pith.science/api/pith-number/BEWL46TUZFEZK5FQB5EXDQFMAP/graph.json","events_json":"https://pith.science/api/pith-number/BEWL46TUZFEZK5FQB5EXDQFMAP/events.json","paper":"https://pith.science/paper/BEWL46TU"},"agent_actions":{"view_html":"https://pith.science/pith/BEWL46TUZFEZK5FQB5EXDQFMAP","download_json":"https://pith.science/pith/BEWL46TUZFEZK5FQB5EXDQFMAP.json","view_paper":"https://pith.science/paper/BEWL46TU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.14662&json=true","fetch_graph":"https://pith.science/api/pith-number/BEWL46TUZFEZK5FQB5EXDQFMAP/graph.json","fetch_events":"https://pith.science/api/pith-number/BEWL46TUZFEZK5FQB5EXDQFMAP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BEWL46TUZFEZK5FQB5EXDQFMAP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BEWL46TUZFEZK5FQB5EXDQFMAP/action/storage_attestation","attest_author":"https://pith.science/pith/BEWL46TUZFEZK5FQB5EXDQFMAP/action/author_attestation","sign_citation":"https://pith.science/pith/BEWL46TUZFEZK5FQB5EXDQFMAP/action/citation_signature","submit_replication":"https://pith.science/pith/BEWL46TUZFEZK5FQB5EXDQFMAP/action/replication_record"}},"created_at":"2026-07-05T08:11:08.291166+00:00","updated_at":"2026-07-05T08:11:08.291166+00:00"}