{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3IQ3PGYEWQV5KYO55WWUDQTORK","short_pith_number":"pith:3IQ3PGYE","schema_version":"1.0","canonical_sha256":"da21b79b04b42bd561ddedad41c26e8aae35fa6254efef72e751be532467971c","source":{"kind":"arxiv","id":"2405.20974","version":3},"attestation_state":"computed","paper":{"title":"SaySelf: Teaching LLMs to Express Confidence with Self-Reflective Rationales","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jing Gao, Shizhe Diao, Shujin Wu, Tianyang Xu, Xiaoze Liu, Xingyao Wang, Yangyi Chen","submitted_at":"2024-05-31T16:21:16Z","abstract_excerpt":"Large language models (LLMs) often generate inaccurate or fabricated information and generally fail to indicate their confidence, which limits their broader applications. Previous work elicits confidence from LLMs by direct or self-consistency prompting, or constructing specific datasets for supervised finetuning. The prompting-based approaches have inferior performance, and the training-based approaches are limited to binary or inaccurate group-level confidence estimates. In this work, we present the advanced SaySelf, a training framework that teaches LLMs to express more accurate fine-graine"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.20974","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-31T16:21:16Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8af9d0004da732e14656f27715efcd5c54021c2b90831419fa89e5a92cb8dda5","abstract_canon_sha256":"1b2f4764045bf4a81f725ed85fa20754f5a5f47962b9751ef44b0c059fd27d05"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:57.982488Z","signature_b64":"syx0c2c3xfOMMOpNir9TBcdVyFj3VGZx6LqPklxUcThYPobHwN0xyfDDmWtHY3twi4r/d/DmgfyyDSlNdOonCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da21b79b04b42bd561ddedad41c26e8aae35fa6254efef72e751be532467971c","last_reissued_at":"2026-07-05T09:15:57.981915Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:57.981915Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SaySelf: Teaching LLMs to Express Confidence with Self-Reflective Rationales","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jing Gao, Shizhe Diao, Shujin Wu, Tianyang Xu, Xiaoze Liu, Xingyao Wang, Yangyi Chen","submitted_at":"2024-05-31T16:21:16Z","abstract_excerpt":"Large language models (LLMs) often generate inaccurate or fabricated information and generally fail to indicate their confidence, which limits their broader applications. Previous work elicits confidence from LLMs by direct or self-consistency prompting, or constructing specific datasets for supervised finetuning. The prompting-based approaches have inferior performance, and the training-based approaches are limited to binary or inaccurate group-level confidence estimates. In this work, we present the advanced SaySelf, a training framework that teaches LLMs to express more accurate fine-graine"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20974","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.20974/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.20974","created_at":"2026-07-05T09:15:57.981990+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.20974v3","created_at":"2026-07-05T09:15:57.981990+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20974","created_at":"2026-07-05T09:15:57.981990+00:00"},{"alias_kind":"pith_short_12","alias_value":"3IQ3PGYEWQV5","created_at":"2026-07-05T09:15:57.981990+00:00"},{"alias_kind":"pith_short_16","alias_value":"3IQ3PGYEWQV5KYO5","created_at":"2026-07-05T09:15:57.981990+00:00"},{"alias_kind":"pith_short_8","alias_value":"3IQ3PGYE","created_at":"2026-07-05T09:15:57.981990+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01612","citing_title":"Scaling with Confidence: Calibrating Confidence of LLMs for Adaptive Test Time Scaling","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30814","citing_title":"When Calibration Rankings Reverse: Accuracy-Controlled Evaluation for Fair Comparison of LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09117","citing_title":"Decoupling Reasoning and Confidence: Resurrecting Calibration in Reinforcement Learning from Verifiable Rewards","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3IQ3PGYEWQV5KYO55WWUDQTORK","json":"https://pith.science/pith/3IQ3PGYEWQV5KYO55WWUDQTORK.json","graph_json":"https://pith.science/api/pith-number/3IQ3PGYEWQV5KYO55WWUDQTORK/graph.json","events_json":"https://pith.science/api/pith-number/3IQ3PGYEWQV5KYO55WWUDQTORK/events.json","paper":"https://pith.science/paper/3IQ3PGYE"},"agent_actions":{"view_html":"https://pith.science/pith/3IQ3PGYEWQV5KYO55WWUDQTORK","download_json":"https://pith.science/pith/3IQ3PGYEWQV5KYO55WWUDQTORK.json","view_paper":"https://pith.science/paper/3IQ3PGYE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.20974&json=true","fetch_graph":"https://pith.science/api/pith-number/3IQ3PGYEWQV5KYO55WWUDQTORK/graph.json","fetch_events":"https://pith.science/api/pith-number/3IQ3PGYEWQV5KYO55WWUDQTORK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3IQ3PGYEWQV5KYO55WWUDQTORK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3IQ3PGYEWQV5KYO55WWUDQTORK/action/storage_attestation","attest_author":"https://pith.science/pith/3IQ3PGYEWQV5KYO55WWUDQTORK/action/author_attestation","sign_citation":"https://pith.science/pith/3IQ3PGYEWQV5KYO55WWUDQTORK/action/citation_signature","submit_replication":"https://pith.science/pith/3IQ3PGYEWQV5KYO55WWUDQTORK/action/replication_record"}},"created_at":"2026-07-05T09:15:57.981990+00:00","updated_at":"2026-07-05T09:15:57.981990+00:00"}