{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B43IVUMLPM7SBQFGGOFPF3VUHF","short_pith_number":"pith:B43IVUML","schema_version":"1.0","canonical_sha256":"0f368ad18b7b3f20c0a6338af2eeb4397917c6c8b264a55230af0e9c1b677ac5","source":{"kind":"arxiv","id":"2501.02901","version":1},"attestation_state":"computed","paper":{"title":"DeCon: Detecting Incorrect Assertions via Postconditions Generated by a Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.PL"],"primary_cat":"cs.SE","authors_text":"Assaf Marron, David Harel, Dezhi Ran, Hao Yu, Jiaming Huang, Tao Xie, Tianyu Chen, Xinyu Wang, Ying Li, Yuan Xie, Zongyang Li","submitted_at":"2025-01-06T10:25:28Z","abstract_excerpt":"Recently, given the docstring for the target problem and the target function signature, large language models (LLMs) have been used not only to generate source code, but also to generate test cases, consisting of test inputs and assertions (e.g., in the form of checking an actual output against the expected output). However, as shown by our empirical study on assertions generated by four LLMs for the HumanEval benchmark, over 62% of the generated assertions are incorrect (i.e., failed on the ground-truth problem solution). To detect incorrect assertions (given the docstring and the target func"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.02901","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2025-01-06T10:25:28Z","cross_cats_sorted":["cs.PL"],"title_canon_sha256":"05a8191ca571c7ecf0af003f460a3f529b8477122cb1c1b8c8096af27a816088","abstract_canon_sha256":"eb465da31fa515eea19b2a44bd032dc5b65188a67173ea5febc0f05b28243477"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:57:26.732045Z","signature_b64":"xJodMY663byYWq+9ouiwnXP4wJk6erkAvSBvCuRBuXGGEXZCMi4N5CB9gxUalCFYinWbFeUnep0f6rqlhUpPAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f368ad18b7b3f20c0a6338af2eeb4397917c6c8b264a55230af0e9c1b677ac5","last_reissued_at":"2026-07-05T09:57:26.731544Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:57:26.731544Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DeCon: Detecting Incorrect Assertions via Postconditions Generated by a Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.PL"],"primary_cat":"cs.SE","authors_text":"Assaf Marron, David Harel, Dezhi Ran, Hao Yu, Jiaming Huang, Tao Xie, Tianyu Chen, Xinyu Wang, Ying Li, Yuan Xie, Zongyang Li","submitted_at":"2025-01-06T10:25:28Z","abstract_excerpt":"Recently, given the docstring for the target problem and the target function signature, large language models (LLMs) have been used not only to generate source code, but also to generate test cases, consisting of test inputs and assertions (e.g., in the form of checking an actual output against the expected output). However, as shown by our empirical study on assertions generated by four LLMs for the HumanEval benchmark, over 62% of the generated assertions are incorrect (i.e., failed on the ground-truth problem solution). To detect incorrect assertions (given the docstring and the target func"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.02901","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.02901/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.02901","created_at":"2026-07-05T09:57:26.731609+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.02901v1","created_at":"2026-07-05T09:57:26.731609+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.02901","created_at":"2026-07-05T09:57:26.731609+00:00"},{"alias_kind":"pith_short_12","alias_value":"B43IVUMLPM7S","created_at":"2026-07-05T09:57:26.731609+00:00"},{"alias_kind":"pith_short_16","alias_value":"B43IVUMLPM7SBQFG","created_at":"2026-07-05T09:57:26.731609+00:00"},{"alias_kind":"pith_short_8","alias_value":"B43IVUML","created_at":"2026-07-05T09:57:26.731609+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.00408","citing_title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B43IVUMLPM7SBQFGGOFPF3VUHF","json":"https://pith.science/pith/B43IVUMLPM7SBQFGGOFPF3VUHF.json","graph_json":"https://pith.science/api/pith-number/B43IVUMLPM7SBQFGGOFPF3VUHF/graph.json","events_json":"https://pith.science/api/pith-number/B43IVUMLPM7SBQFGGOFPF3VUHF/events.json","paper":"https://pith.science/paper/B43IVUML"},"agent_actions":{"view_html":"https://pith.science/pith/B43IVUMLPM7SBQFGGOFPF3VUHF","download_json":"https://pith.science/pith/B43IVUMLPM7SBQFGGOFPF3VUHF.json","view_paper":"https://pith.science/paper/B43IVUML","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.02901&json=true","fetch_graph":"https://pith.science/api/pith-number/B43IVUMLPM7SBQFGGOFPF3VUHF/graph.json","fetch_events":"https://pith.science/api/pith-number/B43IVUMLPM7SBQFGGOFPF3VUHF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B43IVUMLPM7SBQFGGOFPF3VUHF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B43IVUMLPM7SBQFGGOFPF3VUHF/action/storage_attestation","attest_author":"https://pith.science/pith/B43IVUMLPM7SBQFGGOFPF3VUHF/action/author_attestation","sign_citation":"https://pith.science/pith/B43IVUMLPM7SBQFGGOFPF3VUHF/action/citation_signature","submit_replication":"https://pith.science/pith/B43IVUMLPM7SBQFGGOFPF3VUHF/action/replication_record"}},"created_at":"2026-07-05T09:57:26.731609+00:00","updated_at":"2026-07-05T09:57:26.731609+00:00"}