{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2E2QP6PBFLRU4CP4REIQLRPQLW","short_pith_number":"pith:2E2QP6PB","schema_version":"1.0","canonical_sha256":"d13507f9e12ae34e09fc891105c5f05dafeffffbf80bb62f490dfa87a981378c","source":{"kind":"arxiv","id":"2406.11915","version":2},"attestation_state":"computed","paper":{"title":"miniCodeProps: a Minimal Benchmark for Proving Code Properties","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Evan Lohn, Sean Welleck","submitted_at":"2024-06-16T21:11:23Z","abstract_excerpt":"AI agents have shown initial promise in automating mathematical theorem proving in proof assistants such as Lean. The same proof assistants can be used to verify the correctness of code by pairing code with specifications and proofs that the specifications hold. Automating the writing of code, specifications, and proofs could lower the cost of verification, or, ambitiously, enable an AI agent to output safe, provably correct code. However, it remains unclear whether current neural theorem provers can automatically verify even relatively simple programs. We present miniCodeProps, a benchmark of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11915","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2024-06-16T21:11:23Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"41ddef46b90f7a6e8f87ecad5768efc8fd4c58e40c7f4dc3f23cf38e878ea0ae","abstract_canon_sha256":"0670ee98f399c5659721dbd83ad189fc695fc11bdafaacfd91f16e735478ccf2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:18:27.269306Z","signature_b64":"Y45UTV8A6SnoZJBV2Nk2oRstbAxgyZ1gYt/E9XeXpx6L3Btha9XRH8mRGKK4CiHC4lbi0BphuU+V3VZP38WEAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d13507f9e12ae34e09fc891105c5f05dafeffffbf80bb62f490dfa87a981378c","last_reissued_at":"2026-07-05T09:18:27.268829Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:18:27.268829Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"miniCodeProps: a Minimal Benchmark for Proving Code Properties","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Evan Lohn, Sean Welleck","submitted_at":"2024-06-16T21:11:23Z","abstract_excerpt":"AI agents have shown initial promise in automating mathematical theorem proving in proof assistants such as Lean. The same proof assistants can be used to verify the correctness of code by pairing code with specifications and proofs that the specifications hold. Automating the writing of code, specifications, and proofs could lower the cost of verification, or, ambitiously, enable an AI agent to output safe, provably correct code. However, it remains unclear whether current neural theorem provers can automatically verify even relatively simple programs. We present miniCodeProps, a benchmark of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11915","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11915/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11915","created_at":"2026-07-05T09:18:27.268882+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11915v2","created_at":"2026-07-05T09:18:27.268882+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11915","created_at":"2026-07-05T09:18:27.268882+00:00"},{"alias_kind":"pith_short_12","alias_value":"2E2QP6PBFLRU","created_at":"2026-07-05T09:18:27.268882+00:00"},{"alias_kind":"pith_short_16","alias_value":"2E2QP6PBFLRU4CP4","created_at":"2026-07-05T09:18:27.268882+00:00"},{"alias_kind":"pith_short_8","alias_value":"2E2QP6PB","created_at":"2026-07-05T09:18:27.268882+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01008","citing_title":"FVSpec: Real-World Property-Based Tests as Lean Challenges","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23109","citing_title":"Inductive Deductive Synthesis: Enabling AI to Generate Formally Verified Systems","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21104","citing_title":"BRIDGE: Building Representations In Domain Guided Program Synthesis","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14628","citing_title":"s2n-bignum-bench: A practical benchmark for evaluating low-level code reasoning of LLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08553","citing_title":"VeriContest: A Competitive-Programming Benchmark for Verifiable Code Generation","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2E2QP6PBFLRU4CP4REIQLRPQLW","json":"https://pith.science/pith/2E2QP6PBFLRU4CP4REIQLRPQLW.json","graph_json":"https://pith.science/api/pith-number/2E2QP6PBFLRU4CP4REIQLRPQLW/graph.json","events_json":"https://pith.science/api/pith-number/2E2QP6PBFLRU4CP4REIQLRPQLW/events.json","paper":"https://pith.science/paper/2E2QP6PB"},"agent_actions":{"view_html":"https://pith.science/pith/2E2QP6PBFLRU4CP4REIQLRPQLW","download_json":"https://pith.science/pith/2E2QP6PBFLRU4CP4REIQLRPQLW.json","view_paper":"https://pith.science/paper/2E2QP6PB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11915&json=true","fetch_graph":"https://pith.science/api/pith-number/2E2QP6PBFLRU4CP4REIQLRPQLW/graph.json","fetch_events":"https://pith.science/api/pith-number/2E2QP6PBFLRU4CP4REIQLRPQLW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2E2QP6PBFLRU4CP4REIQLRPQLW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2E2QP6PBFLRU4CP4REIQLRPQLW/action/storage_attestation","attest_author":"https://pith.science/pith/2E2QP6PBFLRU4CP4REIQLRPQLW/action/author_attestation","sign_citation":"https://pith.science/pith/2E2QP6PBFLRU4CP4REIQLRPQLW/action/citation_signature","submit_replication":"https://pith.science/pith/2E2QP6PBFLRU4CP4REIQLRPQLW/action/replication_record"}},"created_at":"2026-07-05T09:18:27.268882+00:00","updated_at":"2026-07-05T09:18:27.268882+00:00"}