{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:GDEDYU5IBNFZBFRWYYX4NQV6GO","short_pith_number":"pith:GDEDYU5I","schema_version":"1.0","canonical_sha256":"30c83c53a80b4b909636c62fc6c2be33b480be0158c5975979421e4ef56ba3db","source":{"kind":"arxiv","id":"2606.23196","version":1},"attestation_state":"computed","paper":{"title":"When Does Intrinsic Self-Correction Help? A Task-Sensitive Analysis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dvir Berlowitz, Elroy Stav, Maayan Orner, Sarit Kraus","submitted_at":"2026-06-22T11:44:29Z","abstract_excerpt":"Intrinsic self-correction (SC) aims to improve large language model outputs by prompting a model to revisit its own initial answer without external feedback. Recent studies have questioned the reliability of this approach, showing that models often struggle to judge whether their initial responses are correct. In this work, we take a task-sensitive view of SC. Rather than asking whether it works in general, we examine settings where SC may operate through different mechanisms: verifying explicit constraints, revisiting a complex reasoning process, or providing a second opinion over competing s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2606.23196","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-06-22T11:44:29Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"36dd875147564ef0ebdf39289dd661fb9a3dd6bf76ec4c7628f0ae93dcc5749d","abstract_canon_sha256":"0c299083036bbd0237eb72b8473f2fe0d2ef24da3d32872cff4865f6ced9472a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-23T03:14:12.698844Z","signature_b64":"H/LQQx3C5B5vEIeAtJE+GO7tveQ4Bz9S57GVweSNRWE03c0wVbhi6vwVR3W8Xk/TjOiQ66ZE1E7Qk0aIoNWfAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"30c83c53a80b4b909636c62fc6c2be33b480be0158c5975979421e4ef56ba3db","last_reissued_at":"2026-06-23T03:14:12.698466Z","signature_status":"signed_v1","first_computed_at":"2026-06-23T03:14:12.698466Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Does Intrinsic Self-Correction Help? A Task-Sensitive Analysis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dvir Berlowitz, Elroy Stav, Maayan Orner, Sarit Kraus","submitted_at":"2026-06-22T11:44:29Z","abstract_excerpt":"Intrinsic self-correction (SC) aims to improve large language model outputs by prompting a model to revisit its own initial answer without external feedback. Recent studies have questioned the reliability of this approach, showing that models often struggle to judge whether their initial responses are correct. In this work, we take a task-sensitive view of SC. Rather than asking whether it works in general, we examine settings where SC may operate through different mechanisms: verifying explicit constraints, revisiting a complex reasoning process, or providing a second opinion over competing s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.23196","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2606.23196/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2606.23196","created_at":"2026-06-23T03:14:12.698533+00:00"},{"alias_kind":"arxiv_version","alias_value":"2606.23196v1","created_at":"2026-06-23T03:14:12.698533+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.23196","created_at":"2026-06-23T03:14:12.698533+00:00"},{"alias_kind":"pith_short_12","alias_value":"GDEDYU5IBNFZ","created_at":"2026-06-23T03:14:12.698533+00:00"},{"alias_kind":"pith_short_16","alias_value":"GDEDYU5IBNFZBFRW","created_at":"2026-06-23T03:14:12.698533+00:00"},{"alias_kind":"pith_short_8","alias_value":"GDEDYU5I","created_at":"2026-06-23T03:14:12.698533+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07663","citing_title":"Recursive Self-Improvement in AI: From Bounded Self-Refinement to Autonomous Research Loops","ref_index":30,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GDEDYU5IBNFZBFRWYYX4NQV6GO","json":"https://pith.science/pith/GDEDYU5IBNFZBFRWYYX4NQV6GO.json","graph_json":"https://pith.science/api/pith-number/GDEDYU5IBNFZBFRWYYX4NQV6GO/graph.json","events_json":"https://pith.science/api/pith-number/GDEDYU5IBNFZBFRWYYX4NQV6GO/events.json","paper":"https://pith.science/paper/GDEDYU5I"},"agent_actions":{"view_html":"https://pith.science/pith/GDEDYU5IBNFZBFRWYYX4NQV6GO","download_json":"https://pith.science/pith/GDEDYU5IBNFZBFRWYYX4NQV6GO.json","view_paper":"https://pith.science/paper/GDEDYU5I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2606.23196&json=true","fetch_graph":"https://pith.science/api/pith-number/GDEDYU5IBNFZBFRWYYX4NQV6GO/graph.json","fetch_events":"https://pith.science/api/pith-number/GDEDYU5IBNFZBFRWYYX4NQV6GO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GDEDYU5IBNFZBFRWYYX4NQV6GO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GDEDYU5IBNFZBFRWYYX4NQV6GO/action/storage_attestation","attest_author":"https://pith.science/pith/GDEDYU5IBNFZBFRWYYX4NQV6GO/action/author_attestation","sign_citation":"https://pith.science/pith/GDEDYU5IBNFZBFRWYYX4NQV6GO/action/citation_signature","submit_replication":"https://pith.science/pith/GDEDYU5IBNFZBFRWYYX4NQV6GO/action/replication_record"}},"created_at":"2026-06-23T03:14:12.698533+00:00","updated_at":"2026-06-23T03:14:12.698533+00:00"}