{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FTWHTYKSCXHUXNJZKBO5HFE3SO","short_pith_number":"pith:FTWHTYKS","schema_version":"1.0","canonical_sha256":"2cec79e15215cf4bb539505dd3949b938d5f23f909b36523202d4c660d391d18","source":{"kind":"arxiv","id":"2501.16456","version":3},"attestation_state":"computed","paper":{"title":"CoCoNUT: Structural Code Understanding does not fall out of a tree","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.LG","authors_text":"Claas Beger, Saikat Dutta","submitted_at":"2025-01-27T19:29:11Z","abstract_excerpt":"Large Language Models (LLMs) have shown impressive performance across a wide array of tasks involving both structured and unstructured textual data. Recent results on various benchmarks for code generation, repair, or completion suggest that certain models have programming abilities comparable to or even surpass humans. In this work, we demonstrate that high performance on such benchmarks does not correlate to humans' innate ability to understand structural control flow in code. To this end, we extract solutions from the HumanEval benchmark, which the relevant models perform strongly on, and t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16456","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-27T19:29:11Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"93357199adc8e3eb5c95fdb97e6b67e8442c72e239994d236a608ea99f6527cc","abstract_canon_sha256":"c6d5f64d59b0e7de01a2dfd3494e1861a0f366a277f97ce448d29376c44d7569"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:23:28.812520Z","signature_b64":"LsPHskRZGq55JTZkeW0q3DxBaQfQDqrCethgyguK3TglnUc9G94VCOR1d3ViGHc+PBv5qonlx4XG+WxVCBVoCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2cec79e15215cf4bb539505dd3949b938d5f23f909b36523202d4c660d391d18","last_reissued_at":"2026-07-05T10:23:28.812030Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:23:28.812030Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CoCoNUT: Structural Code Understanding does not fall out of a tree","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.LG","authors_text":"Claas Beger, Saikat Dutta","submitted_at":"2025-01-27T19:29:11Z","abstract_excerpt":"Large Language Models (LLMs) have shown impressive performance across a wide array of tasks involving both structured and unstructured textual data. Recent results on various benchmarks for code generation, repair, or completion suggest that certain models have programming abilities comparable to or even surpass humans. In this work, we demonstrate that high performance on such benchmarks does not correlate to humans' innate ability to understand structural control flow in code. To this end, we extract solutions from the HumanEval benchmark, which the relevant models perform strongly on, and t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16456","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16456/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16456","created_at":"2026-07-05T10:23:28.812089+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16456v3","created_at":"2026-07-05T10:23:28.812089+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16456","created_at":"2026-07-05T10:23:28.812089+00:00"},{"alias_kind":"pith_short_12","alias_value":"FTWHTYKSCXHU","created_at":"2026-07-05T10:23:28.812089+00:00"},{"alias_kind":"pith_short_16","alias_value":"FTWHTYKSCXHUXNJZ","created_at":"2026-07-05T10:23:28.812089+00:00"},{"alias_kind":"pith_short_8","alias_value":"FTWHTYKS","created_at":"2026-07-05T10:23:28.812089+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2402.09664","citing_title":"CodeMind: Evaluating Large Language Models for Code Reasoning","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15079","citing_title":"Assessing Coherency and Consistency of Code Execution Reasoning by Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14917","citing_title":"Evaluating Code Reasoning Abilities of Large Language Models Under Real-World Settings","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FTWHTYKSCXHUXNJZKBO5HFE3SO","json":"https://pith.science/pith/FTWHTYKSCXHUXNJZKBO5HFE3SO.json","graph_json":"https://pith.science/api/pith-number/FTWHTYKSCXHUXNJZKBO5HFE3SO/graph.json","events_json":"https://pith.science/api/pith-number/FTWHTYKSCXHUXNJZKBO5HFE3SO/events.json","paper":"https://pith.science/paper/FTWHTYKS"},"agent_actions":{"view_html":"https://pith.science/pith/FTWHTYKSCXHUXNJZKBO5HFE3SO","download_json":"https://pith.science/pith/FTWHTYKSCXHUXNJZKBO5HFE3SO.json","view_paper":"https://pith.science/paper/FTWHTYKS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16456&json=true","fetch_graph":"https://pith.science/api/pith-number/FTWHTYKSCXHUXNJZKBO5HFE3SO/graph.json","fetch_events":"https://pith.science/api/pith-number/FTWHTYKSCXHUXNJZKBO5HFE3SO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FTWHTYKSCXHUXNJZKBO5HFE3SO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FTWHTYKSCXHUXNJZKBO5HFE3SO/action/storage_attestation","attest_author":"https://pith.science/pith/FTWHTYKSCXHUXNJZKBO5HFE3SO/action/author_attestation","sign_citation":"https://pith.science/pith/FTWHTYKSCXHUXNJZKBO5HFE3SO/action/citation_signature","submit_replication":"https://pith.science/pith/FTWHTYKSCXHUXNJZKBO5HFE3SO/action/replication_record"}},"created_at":"2026-07-05T10:23:28.812089+00:00","updated_at":"2026-07-05T10:23:28.812089+00:00"}