{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BM2XGT3DBNOHR47DFL757H3SFJ","short_pith_number":"pith:BM2XGT3D","schema_version":"1.0","canonical_sha256":"0b35734f630b5c78f3e32affdf9f722a5f3b0f94f8b7c8a20e2899720d160f1f","source":{"kind":"arxiv","id":"2504.02902","version":1},"attestation_state":"computed","paper":{"title":"Beyond Accuracy: The Role of Calibration in Self-Improving Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dawei Li, Huan Liu, Liangjie Huang, Lu Cheng","submitted_at":"2025-04-03T04:39:54Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable self-improvement capabilities, whereby models iteratively revise their outputs through self-generated feedback. While this reflective mechanism has shown promise in enhancing task performance, recent studies suggest that it may also introduce undesirable biases-most notably, self-bias, or the tendency of LLMs to favor their own prior outputs. In this work, we extend this line of inquiry by investigating the impact on confidence estimation. We evaluate three representative self-improvement paradigms-basic prompting, Chain-of-Thought (CoT"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.02902","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-03T04:39:54Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"502e14d412c5feb48f5599f811091bfcece4271c27518c8274f75c94fe427a47","abstract_canon_sha256":"2d586bc149010074ab9a9d403119818c389642be32b5cd06a1cbaacce6355102"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:44:04.877355Z","signature_b64":"8qVkdYF8+Aat3tWKQPC6aRVWnN/lNWUTDAUmcM2b0gjpn8NJx4GmTy1IJD/fS8+8l+sj7oWa2KFS4vtce4LGCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0b35734f630b5c78f3e32affdf9f722a5f3b0f94f8b7c8a20e2899720d160f1f","last_reissued_at":"2026-07-05T10:44:04.876842Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:44:04.876842Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Accuracy: The Role of Calibration in Self-Improving Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dawei Li, Huan Liu, Liangjie Huang, Lu Cheng","submitted_at":"2025-04-03T04:39:54Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable self-improvement capabilities, whereby models iteratively revise their outputs through self-generated feedback. While this reflective mechanism has shown promise in enhancing task performance, recent studies suggest that it may also introduce undesirable biases-most notably, self-bias, or the tendency of LLMs to favor their own prior outputs. In this work, we extend this line of inquiry by investigating the impact on confidence estimation. We evaluate three representative self-improvement paradigms-basic prompting, Chain-of-Thought (CoT"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.02902","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.02902/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.02902","created_at":"2026-07-05T10:44:04.876904+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.02902v1","created_at":"2026-07-05T10:44:04.876904+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.02902","created_at":"2026-07-05T10:44:04.876904+00:00"},{"alias_kind":"pith_short_12","alias_value":"BM2XGT3DBNOH","created_at":"2026-07-05T10:44:04.876904+00:00"},{"alias_kind":"pith_short_16","alias_value":"BM2XGT3DBNOHR47D","created_at":"2026-07-05T10:44:04.876904+00:00"},{"alias_kind":"pith_short_8","alias_value":"BM2XGT3D","created_at":"2026-07-05T10:44:04.876904+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07361","citing_title":"BUS: Brain-Inspired Unsupervised Self-Reflection via Backward Prediction for Multimodal Reasoning","ref_index":83,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01702","citing_title":"Pmeta-TLA: Backdoor Attacks for Speech Classification Models via Meta-Learning with Timbre Leakage Attack","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BM2XGT3DBNOHR47DFL757H3SFJ","json":"https://pith.science/pith/BM2XGT3DBNOHR47DFL757H3SFJ.json","graph_json":"https://pith.science/api/pith-number/BM2XGT3DBNOHR47DFL757H3SFJ/graph.json","events_json":"https://pith.science/api/pith-number/BM2XGT3DBNOHR47DFL757H3SFJ/events.json","paper":"https://pith.science/paper/BM2XGT3D"},"agent_actions":{"view_html":"https://pith.science/pith/BM2XGT3DBNOHR47DFL757H3SFJ","download_json":"https://pith.science/pith/BM2XGT3DBNOHR47DFL757H3SFJ.json","view_paper":"https://pith.science/paper/BM2XGT3D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.02902&json=true","fetch_graph":"https://pith.science/api/pith-number/BM2XGT3DBNOHR47DFL757H3SFJ/graph.json","fetch_events":"https://pith.science/api/pith-number/BM2XGT3DBNOHR47DFL757H3SFJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BM2XGT3DBNOHR47DFL757H3SFJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BM2XGT3DBNOHR47DFL757H3SFJ/action/storage_attestation","attest_author":"https://pith.science/pith/BM2XGT3DBNOHR47DFL757H3SFJ/action/author_attestation","sign_citation":"https://pith.science/pith/BM2XGT3DBNOHR47DFL757H3SFJ/action/citation_signature","submit_replication":"https://pith.science/pith/BM2XGT3DBNOHR47DFL757H3SFJ/action/replication_record"}},"created_at":"2026-07-05T10:44:04.876904+00:00","updated_at":"2026-07-05T10:44:04.876904+00:00"}