{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EWAPPLTEAGVPB7VLA4DMHATZ5W","short_pith_number":"pith:EWAPPLTE","schema_version":"1.0","canonical_sha256":"2580f7ae6401aaf0feab0706c38279ed800a29c51cdf4530b5fba7a39086bc72","source":{"kind":"arxiv","id":"2503.19599","version":2},"attestation_state":"computed","paper":{"title":"HoarePrompt: Structural Reasoning About Program Correctness in Natural Language","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Dimitrios Stamatios Bouras, Sergey Mechtaev, Tairan Wang, Yihan Dai, Yingfei Xiong","submitted_at":"2025-03-25T12:30:30Z","abstract_excerpt":"While software requirements are often expressed in natural language, verifying the correctness of a program against such requirements is a hard and underexplored problem. Large language models (LLMs) are promising candidates for addressing this challenge, however our experience shows that they are ineffective in this task, often failing to detect even straightforward bugs. To address this gap, we introduce HoarePrompt, a novel approach that adapts fundamental ideas from program verification to natural language artifacts. Inspired from the strongest postcondition calculus, HoarePrompt employs a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.19599","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-03-25T12:30:30Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7f324666c320615dc08fe413d18cfd7198e6add2f27bb2c887a136112bf55d95","abstract_canon_sha256":"e3d1310a64ed99967e9b8f1b0bcabd51d454591da55e9e5ea6747c1ee4a69582"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:57:55.462859Z","signature_b64":"OiK6N2jvzWF85cULDbhNHHtYfshAwaR6PFw7EsO9VxpkmqSHAj1SbcDt7JxvOrzJSaey3VlfXJFPLSJ2tj6wDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2580f7ae6401aaf0feab0706c38279ed800a29c51cdf4530b5fba7a39086bc72","last_reissued_at":"2026-07-05T11:57:55.462377Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:57:55.462377Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HoarePrompt: Structural Reasoning About Program Correctness in Natural Language","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Dimitrios Stamatios Bouras, Sergey Mechtaev, Tairan Wang, Yihan Dai, Yingfei Xiong","submitted_at":"2025-03-25T12:30:30Z","abstract_excerpt":"While software requirements are often expressed in natural language, verifying the correctness of a program against such requirements is a hard and underexplored problem. Large language models (LLMs) are promising candidates for addressing this challenge, however our experience shows that they are ineffective in this task, often failing to detect even straightforward bugs. To address this gap, we introduce HoarePrompt, a novel approach that adapts fundamental ideas from program verification to natural language artifacts. Inspired from the strongest postcondition calculus, HoarePrompt employs a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.19599","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.19599/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.19599","created_at":"2026-07-05T11:57:55.462434+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.19599v2","created_at":"2026-07-05T11:57:55.462434+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.19599","created_at":"2026-07-05T11:57:55.462434+00:00"},{"alias_kind":"pith_short_12","alias_value":"EWAPPLTEAGVP","created_at":"2026-07-05T11:57:55.462434+00:00"},{"alias_kind":"pith_short_16","alias_value":"EWAPPLTEAGVPB7VL","created_at":"2026-07-05T11:57:55.462434+00:00"},{"alias_kind":"pith_short_8","alias_value":"EWAPPLTE","created_at":"2026-07-05T11:57:55.462434+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01903","citing_title":"Rethinking Complexity Metrics for LLM-Integrated Applications: Beyond Source Code","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29822","citing_title":"Inferring Code Correctness from Specification","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14917","citing_title":"Evaluating Code Reasoning Abilities of Large Language Models Under Real-World Settings","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14972","citing_title":"Viverra: Text-to-Code with Guarantees","ref_index":78,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EWAPPLTEAGVPB7VLA4DMHATZ5W","json":"https://pith.science/pith/EWAPPLTEAGVPB7VLA4DMHATZ5W.json","graph_json":"https://pith.science/api/pith-number/EWAPPLTEAGVPB7VLA4DMHATZ5W/graph.json","events_json":"https://pith.science/api/pith-number/EWAPPLTEAGVPB7VLA4DMHATZ5W/events.json","paper":"https://pith.science/paper/EWAPPLTE"},"agent_actions":{"view_html":"https://pith.science/pith/EWAPPLTEAGVPB7VLA4DMHATZ5W","download_json":"https://pith.science/pith/EWAPPLTEAGVPB7VLA4DMHATZ5W.json","view_paper":"https://pith.science/paper/EWAPPLTE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.19599&json=true","fetch_graph":"https://pith.science/api/pith-number/EWAPPLTEAGVPB7VLA4DMHATZ5W/graph.json","fetch_events":"https://pith.science/api/pith-number/EWAPPLTEAGVPB7VLA4DMHATZ5W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EWAPPLTEAGVPB7VLA4DMHATZ5W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EWAPPLTEAGVPB7VLA4DMHATZ5W/action/storage_attestation","attest_author":"https://pith.science/pith/EWAPPLTEAGVPB7VLA4DMHATZ5W/action/author_attestation","sign_citation":"https://pith.science/pith/EWAPPLTEAGVPB7VLA4DMHATZ5W/action/citation_signature","submit_replication":"https://pith.science/pith/EWAPPLTEAGVPB7VLA4DMHATZ5W/action/replication_record"}},"created_at":"2026-07-05T11:57:55.462434+00:00","updated_at":"2026-07-05T11:57:55.462434+00:00"}