{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ETGSVAEXMUCPFBR6JQQ3A72MPT","short_pith_number":"pith:ETGSVAEX","schema_version":"1.0","canonical_sha256":"24cd2a80976504f2863e4c21b07f4c7cd4d0acbfce6861269d141c06474ccb48","source":{"kind":"arxiv","id":"2310.12397","version":1},"attestation_state":"computed","paper":{"title":"GPT-4 Doesn't Know It's Wrong: An Analysis of Iterative Prompting for Reasoning Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Kaya Stechly, Matthew Marquez, Subbarao Kambhampati","submitted_at":"2023-10-19T00:56:37Z","abstract_excerpt":"There has been considerable divergence of opinion on the reasoning abilities of Large Language Models (LLMs). While the initial optimism that reasoning might emerge automatically with scale has been tempered thanks to a slew of counterexamples, a wide spread belief in their iterative self-critique capabilities persists. In this paper, we set out to systematically investigate the effectiveness of iterative prompting of LLMs in the context of Graph Coloring, a canonical NP-complete reasoning problem that is related to propositional satisfiability as well as practical problems like scheduling and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.12397","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-10-19T00:56:37Z","cross_cats_sorted":[],"title_canon_sha256":"74c88a3af07326384507b07cffcd06600afd4ba37d0b4f1e3ab0f5983bc4fceb","abstract_canon_sha256":"01ad3221712f71d9fc428f83e2f8caff049e7da37165481a87fd88ae9aad2421"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:02:37.402936Z","signature_b64":"G7FeoR/oWPQj8L67MZoGJYRqAjaUpYjLQoepduobW1SdRZyxsuLY/VvAu5Jp2Lq9sq2qSmUguZksGrw4IN9PCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"24cd2a80976504f2863e4c21b07f4c7cd4d0acbfce6861269d141c06474ccb48","last_reissued_at":"2026-07-05T07:02:37.402438Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:02:37.402438Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GPT-4 Doesn't Know It's Wrong: An Analysis of Iterative Prompting for Reasoning Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Kaya Stechly, Matthew Marquez, Subbarao Kambhampati","submitted_at":"2023-10-19T00:56:37Z","abstract_excerpt":"There has been considerable divergence of opinion on the reasoning abilities of Large Language Models (LLMs). While the initial optimism that reasoning might emerge automatically with scale has been tempered thanks to a slew of counterexamples, a wide spread belief in their iterative self-critique capabilities persists. In this paper, we set out to systematically investigate the effectiveness of iterative prompting of LLMs in the context of Graph Coloring, a canonical NP-complete reasoning problem that is related to propositional satisfiability as well as practical problems like scheduling and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.12397","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.12397/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.12397","created_at":"2026-07-05T07:02:37.402503+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.12397v1","created_at":"2026-07-05T07:02:37.402503+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.12397","created_at":"2026-07-05T07:02:37.402503+00:00"},{"alias_kind":"pith_short_12","alias_value":"ETGSVAEXMUCP","created_at":"2026-07-05T07:02:37.402503+00:00"},{"alias_kind":"pith_short_16","alias_value":"ETGSVAEXMUCPFBR6","created_at":"2026-07-05T07:02:37.402503+00:00"},{"alias_kind":"pith_short_8","alias_value":"ETGSVAEX","created_at":"2026-07-05T07:02:37.402503+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08068","citing_title":"DICE: Entropy-Regularized Equilibrium Selection for Stable Multi-Agent LLM Coordination","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31404","citing_title":"The Sword, Shield, and Achilles' Heel: Characterizing the Linguistic Inductive Bias of Large Language Models for Spatial Reasoning in Navigation Planning","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05976","citing_title":"The Self-Correction Illusion: Role Relabeling Gates Explicit Error Flagging in Large Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16941","citing_title":"Roll Out and Roll Back: Diffusion LLMs are Their Own Efficiency Teachers","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08904","citing_title":"OPT-BENCH: Evaluating the Iterative Self-Optimization of LLM Agents in Large-Scale Search Spaces","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ETGSVAEXMUCPFBR6JQQ3A72MPT","json":"https://pith.science/pith/ETGSVAEXMUCPFBR6JQQ3A72MPT.json","graph_json":"https://pith.science/api/pith-number/ETGSVAEXMUCPFBR6JQQ3A72MPT/graph.json","events_json":"https://pith.science/api/pith-number/ETGSVAEXMUCPFBR6JQQ3A72MPT/events.json","paper":"https://pith.science/paper/ETGSVAEX"},"agent_actions":{"view_html":"https://pith.science/pith/ETGSVAEXMUCPFBR6JQQ3A72MPT","download_json":"https://pith.science/pith/ETGSVAEXMUCPFBR6JQQ3A72MPT.json","view_paper":"https://pith.science/paper/ETGSVAEX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.12397&json=true","fetch_graph":"https://pith.science/api/pith-number/ETGSVAEXMUCPFBR6JQQ3A72MPT/graph.json","fetch_events":"https://pith.science/api/pith-number/ETGSVAEXMUCPFBR6JQQ3A72MPT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ETGSVAEXMUCPFBR6JQQ3A72MPT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ETGSVAEXMUCPFBR6JQQ3A72MPT/action/storage_attestation","attest_author":"https://pith.science/pith/ETGSVAEXMUCPFBR6JQQ3A72MPT/action/author_attestation","sign_citation":"https://pith.science/pith/ETGSVAEXMUCPFBR6JQQ3A72MPT/action/citation_signature","submit_replication":"https://pith.science/pith/ETGSVAEXMUCPFBR6JQQ3A72MPT/action/replication_record"}},"created_at":"2026-07-05T07:02:37.402503+00:00","updated_at":"2026-07-05T07:02:37.402503+00:00"}