{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:A7MRO3DUPPN5AVWWAUHP4DNQNY","short_pith_number":"pith:A7MRO3DU","schema_version":"1.0","canonical_sha256":"07d9176c747bdbd056d6050efe0db06e250d63217c6e32c41654842221ea6c21","source":{"kind":"arxiv","id":"2401.15963","version":3},"attestation_state":"computed","paper":{"title":"NoFunEval: Funny How Code LMs Falter on Requirements Beyond Functional Correctness","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Abhijeet Awasthi, Aditya Kanade, Manav Singhal, Nagarajan Natarajan, Tushar Aggarwal","submitted_at":"2024-01-29T08:47:31Z","abstract_excerpt":"Existing evaluation benchmarks of language models of code (code LMs) focus almost exclusively on whether the LMs can generate functionally-correct code. In real-world software engineering, developers think beyond functional correctness. They have requirements on \"how\" a functionality should be implemented to meet overall system design objectives like efficiency, security, and maintainability. They would also trust the code LMs more if the LMs demonstrate robust understanding of such requirements.\n  We propose a new benchmark NoFunEval to evaluate code LMs on non-functional requirements and sim"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.15963","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-01-29T08:47:31Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"b0d357a1c040c77129c37f570f1ecd5d5ada4b5d4a0bb2421610856b0a2a04b0","abstract_canon_sha256":"f405e36e78791f2cfd49ec7a2e5b71cf2881785ca44ed2ca3ce521a5f239e174"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:58.974421Z","signature_b64":"m4cNoO+OCPt0vw+oZ7VdZTtlhDkXD0hwfVHDUBkGy+A4xRNLlZDI0Rbm0RFH1H6npmg4Y8imK8laNUPrC6lBAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07d9176c747bdbd056d6050efe0db06e250d63217c6e32c41654842221ea6c21","last_reissued_at":"2026-07-05T09:12:58.973941Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:58.973941Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NoFunEval: Funny How Code LMs Falter on Requirements Beyond Functional Correctness","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Abhijeet Awasthi, Aditya Kanade, Manav Singhal, Nagarajan Natarajan, Tushar Aggarwal","submitted_at":"2024-01-29T08:47:31Z","abstract_excerpt":"Existing evaluation benchmarks of language models of code (code LMs) focus almost exclusively on whether the LMs can generate functionally-correct code. In real-world software engineering, developers think beyond functional correctness. They have requirements on \"how\" a functionality should be implemented to meet overall system design objectives like efficiency, security, and maintainability. They would also trust the code LMs more if the LMs demonstrate robust understanding of such requirements.\n  We propose a new benchmark NoFunEval to evaluate code LMs on non-functional requirements and sim"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.15963","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.15963/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.15963","created_at":"2026-07-05T09:12:58.973997+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.15963v3","created_at":"2026-07-05T09:12:58.973997+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.15963","created_at":"2026-07-05T09:12:58.973997+00:00"},{"alias_kind":"pith_short_12","alias_value":"A7MRO3DUPPN5","created_at":"2026-07-05T09:12:58.973997+00:00"},{"alias_kind":"pith_short_16","alias_value":"A7MRO3DUPPN5AVWW","created_at":"2026-07-05T09:12:58.973997+00:00"},{"alias_kind":"pith_short_8","alias_value":"A7MRO3DU","created_at":"2026-07-05T09:12:58.973997+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07619","citing_title":"Rethinking Code Performance Benchmarks for LLMs","ref_index":36,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31767","citing_title":"JETO-Bench: A Reproducible Benchmark for Execution Time Improvement Patches in Java","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28998","citing_title":"Reward-Free Code Alignment from Pretrained or Fine-Tuned LLM: Unpacking the Trade-offs for Code Generation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17338","citing_title":"Precise Debugging Benchmark: Is Your Model Debugging or Regenerating?","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":233,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17338","citing_title":"Precise Debugging Benchmark: Is Your Model Debugging or Regenerating?","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A7MRO3DUPPN5AVWWAUHP4DNQNY","json":"https://pith.science/pith/A7MRO3DUPPN5AVWWAUHP4DNQNY.json","graph_json":"https://pith.science/api/pith-number/A7MRO3DUPPN5AVWWAUHP4DNQNY/graph.json","events_json":"https://pith.science/api/pith-number/A7MRO3DUPPN5AVWWAUHP4DNQNY/events.json","paper":"https://pith.science/paper/A7MRO3DU"},"agent_actions":{"view_html":"https://pith.science/pith/A7MRO3DUPPN5AVWWAUHP4DNQNY","download_json":"https://pith.science/pith/A7MRO3DUPPN5AVWWAUHP4DNQNY.json","view_paper":"https://pith.science/paper/A7MRO3DU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.15963&json=true","fetch_graph":"https://pith.science/api/pith-number/A7MRO3DUPPN5AVWWAUHP4DNQNY/graph.json","fetch_events":"https://pith.science/api/pith-number/A7MRO3DUPPN5AVWWAUHP4DNQNY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A7MRO3DUPPN5AVWWAUHP4DNQNY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A7MRO3DUPPN5AVWWAUHP4DNQNY/action/storage_attestation","attest_author":"https://pith.science/pith/A7MRO3DUPPN5AVWWAUHP4DNQNY/action/author_attestation","sign_citation":"https://pith.science/pith/A7MRO3DUPPN5AVWWAUHP4DNQNY/action/citation_signature","submit_replication":"https://pith.science/pith/A7MRO3DUPPN5AVWWAUHP4DNQNY/action/replication_record"}},"created_at":"2026-07-05T09:12:58.973997+00:00","updated_at":"2026-07-05T09:12:58.973997+00:00"}