{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IPYWYWZNEXD6KOCEN75QMJRWLV","short_pith_number":"pith:IPYWYWZN","schema_version":"1.0","canonical_sha256":"43f16c5b2d25c7e538446ffb0626365d5cd7ad88c663f68d166823e06c25c119","source":{"kind":"arxiv","id":"2402.08115","version":2},"attestation_state":"computed","paper":{"title":"On the Self-Verification Limitations of Large Language Models on Reasoning and Planning Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Karthik Valmeekam, Kaya Stechly, Subbarao Kambhampati","submitted_at":"2024-02-12T23:11:01Z","abstract_excerpt":"There has been considerable divergence of opinion on the reasoning abilities of Large Language Models (LLMs). While the initial optimism that reasoning might emerge automatically with scale has been tempered thanks to a slew of counterexamples--ranging from multiplication to simple planning--there persists a wide spread belief that LLMs can self-critique and improve their own solutions in an iterative fashion. This belief seemingly rests on the assumption that verification of correctness should be easier than generation--a rather classical argument from computational complexity--which should b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.08115","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-02-12T23:11:01Z","cross_cats_sorted":[],"title_canon_sha256":"674f1e1476399bb83c22ff4681988349cbd8343de91f6cea0b35b42dbc3debad","abstract_canon_sha256":"c47da14821bde0eae13c996f477c47ee02966b4805a9e0a06047bc4edec57d10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:51:47.147146Z","signature_b64":"oNIebHL0NdlC1RdL315uYQczYm7SMOfmrKaHv10azjw/L3sWcmedR959PRxeLEXQCXUA6+OMhkV9EPTc8lVEDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43f16c5b2d25c7e538446ffb0626365d5cd7ad88c663f68d166823e06c25c119","last_reissued_at":"2026-07-05T08:51:47.146704Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:51:47.146704Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Self-Verification Limitations of Large Language Models on Reasoning and Planning Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Karthik Valmeekam, Kaya Stechly, Subbarao Kambhampati","submitted_at":"2024-02-12T23:11:01Z","abstract_excerpt":"There has been considerable divergence of opinion on the reasoning abilities of Large Language Models (LLMs). While the initial optimism that reasoning might emerge automatically with scale has been tempered thanks to a slew of counterexamples--ranging from multiplication to simple planning--there persists a wide spread belief that LLMs can self-critique and improve their own solutions in an iterative fashion. This belief seemingly rests on the assumption that verification of correctness should be easier than generation--a rather classical argument from computational complexity--which should b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.08115","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.08115/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.08115","created_at":"2026-07-05T08:51:47.146759+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.08115v2","created_at":"2026-07-05T08:51:47.146759+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.08115","created_at":"2026-07-05T08:51:47.146759+00:00"},{"alias_kind":"pith_short_12","alias_value":"IPYWYWZNEXD6","created_at":"2026-07-05T08:51:47.146759+00:00"},{"alias_kind":"pith_short_16","alias_value":"IPYWYWZNEXD6KOCE","created_at":"2026-07-05T08:51:47.146759+00:00"},{"alias_kind":"pith_short_8","alias_value":"IPYWYWZN","created_at":"2026-07-05T08:51:47.146759+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06873","citing_title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","ref_index":57,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31511","citing_title":"Falsification, Not Exposure: An Internally Preregistered Placebo-Controlled Decomposition of Self-Repair Feedback in Frozen Small Code Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31399","citing_title":"World-Model Collapse as a Phase Transition","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18933","citing_title":"Zero-Shot Active Feature Acquisition via LLM-Elicitation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2504.01990","citing_title":"Advances and Challenges in Foundation Agents: From Brain-Inspired Intelligence to Evolutionary, Collaborative, and Safe Systems","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17041","citing_title":"Agentic AI Translate: An Agentic Translator Prototype for Translation as Communication Design","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15513","citing_title":"CAPS: Cascaded Adaptive Pairwise Selection for Efficient Parallel Reasoning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15134","citing_title":"The Unreasonable Effectiveness of Entropy Minimization in LLM Reasoning","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14051","citing_title":"SPIN: Structural LLM Planning via Iterative Navigation for Industrial Tasks","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09519","citing_title":"Weighted Rules under the Stable Model Semantics","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02765","citing_title":"U-Define: Designing User Workflows for Hard and Soft Constraints in LLM-Based Planning","ref_index":110,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IPYWYWZNEXD6KOCEN75QMJRWLV","json":"https://pith.science/pith/IPYWYWZNEXD6KOCEN75QMJRWLV.json","graph_json":"https://pith.science/api/pith-number/IPYWYWZNEXD6KOCEN75QMJRWLV/graph.json","events_json":"https://pith.science/api/pith-number/IPYWYWZNEXD6KOCEN75QMJRWLV/events.json","paper":"https://pith.science/paper/IPYWYWZN"},"agent_actions":{"view_html":"https://pith.science/pith/IPYWYWZNEXD6KOCEN75QMJRWLV","download_json":"https://pith.science/pith/IPYWYWZNEXD6KOCEN75QMJRWLV.json","view_paper":"https://pith.science/paper/IPYWYWZN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.08115&json=true","fetch_graph":"https://pith.science/api/pith-number/IPYWYWZNEXD6KOCEN75QMJRWLV/graph.json","fetch_events":"https://pith.science/api/pith-number/IPYWYWZNEXD6KOCEN75QMJRWLV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IPYWYWZNEXD6KOCEN75QMJRWLV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IPYWYWZNEXD6KOCEN75QMJRWLV/action/storage_attestation","attest_author":"https://pith.science/pith/IPYWYWZNEXD6KOCEN75QMJRWLV/action/author_attestation","sign_citation":"https://pith.science/pith/IPYWYWZNEXD6KOCEN75QMJRWLV/action/citation_signature","submit_replication":"https://pith.science/pith/IPYWYWZNEXD6KOCEN75QMJRWLV/action/replication_record"}},"created_at":"2026-07-05T08:51:47.146759+00:00","updated_at":"2026-07-05T08:51:47.146759+00:00"}