{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HISRIZNZQDMZIGTIE5EFOEQ3SB","short_pith_number":"pith:HISRIZNZ","schema_version":"1.0","canonical_sha256":"3a251465b980d9941a68274857121b904eb0bdb462a7f9854fb77a8b3f0f100a","source":{"kind":"arxiv","id":"2312.07000","version":2},"attestation_state":"computed","paper":{"title":"Alignment for Honesty","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ethan Chern, Graham Neubig, Pengfei Liu, Xipeng Qiu, Yuqing Yang","submitted_at":"2023-12-12T06:10:42Z","abstract_excerpt":"Recent research has made significant strides in aligning large language models (LLMs) with helpfulness and harmlessness. In this paper, we argue for the importance of alignment for \\emph{honesty}, ensuring that LLMs proactively refuse to answer questions when they lack knowledge, while still not being overly conservative. However, a pivotal aspect of alignment for honesty involves discerning an LLM's knowledge boundaries, which demands comprehensive solutions in terms of metric development, benchmark creation, and training methodologies. We address these challenges by first establishing a prec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.07000","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-12T06:10:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7d866dab8f182a90948bc12456b3386e18fa8e988567303bed4b548ec5f07d94","abstract_canon_sha256":"b85129e422a59c958cc2dedaba21d08198a90bb423d92972ef6a250de4af3917"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:26:57.462579Z","signature_b64":"THmM4W/GWI+wviTqu7oCsx2t/u+Jr/WJrljeNjHZplxAGjd0UaBkkSxNrUWyaQy5ZY8unU2DASysDabfh8jYDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a251465b980d9941a68274857121b904eb0bdb462a7f9854fb77a8b3f0f100a","last_reissued_at":"2026-07-05T09:26:57.462016Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:26:57.462016Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Alignment for Honesty","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ethan Chern, Graham Neubig, Pengfei Liu, Xipeng Qiu, Yuqing Yang","submitted_at":"2023-12-12T06:10:42Z","abstract_excerpt":"Recent research has made significant strides in aligning large language models (LLMs) with helpfulness and harmlessness. In this paper, we argue for the importance of alignment for \\emph{honesty}, ensuring that LLMs proactively refuse to answer questions when they lack knowledge, while still not being overly conservative. However, a pivotal aspect of alignment for honesty involves discerning an LLM's knowledge boundaries, which demands comprehensive solutions in terms of metric development, benchmark creation, and training methodologies. We address these challenges by first establishing a prec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.07000","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.07000/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.07000","created_at":"2026-07-05T09:26:57.462083+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.07000v2","created_at":"2026-07-05T09:26:57.462083+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.07000","created_at":"2026-07-05T09:26:57.462083+00:00"},{"alias_kind":"pith_short_12","alias_value":"HISRIZNZQDMZ","created_at":"2026-07-05T09:26:57.462083+00:00"},{"alias_kind":"pith_short_16","alias_value":"HISRIZNZQDMZIGTI","created_at":"2026-07-05T09:26:57.462083+00:00"},{"alias_kind":"pith_short_8","alias_value":"HISRIZNZ","created_at":"2026-07-05T09:26:57.462083+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08399","citing_title":"Prompt Compression via Activation Aggregation","ref_index":63,"is_internal_anchor":true},{"citing_arxiv_id":"2606.13310","citing_title":"RogueAI: A Reverse Turing Test for Detecting Licensed AI Deception in Dialogue","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11712","citing_title":"Substrate Asymmetry in User-Side Memory: A Diagnostic Framework","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03535","citing_title":"Can LLM Rerankers Predict Their Own Ranking Performance?","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32032","citing_title":"Reinforcement Learning with Metacognitive Feedback Elicits Faithful Uncertainty Expression in LLMs","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28778","citing_title":"Can LLMs Use Linguistic Uncertainty Markers to Reliably Reflect Intrinsic Confidence?","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06315","citing_title":"LLM Self-Recognition: Steering and Retrieving Activation Signatures","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2402.13116","citing_title":"A Survey on Knowledge Distillation of Large Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10391","citing_title":"Phoenix-VL 1.5 Medium Technical Report","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17200","citing_title":"Calibrating Model-Based Evaluation Metrics for Summarization","ref_index":70,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HISRIZNZQDMZIGTIE5EFOEQ3SB","json":"https://pith.science/pith/HISRIZNZQDMZIGTIE5EFOEQ3SB.json","graph_json":"https://pith.science/api/pith-number/HISRIZNZQDMZIGTIE5EFOEQ3SB/graph.json","events_json":"https://pith.science/api/pith-number/HISRIZNZQDMZIGTIE5EFOEQ3SB/events.json","paper":"https://pith.science/paper/HISRIZNZ"},"agent_actions":{"view_html":"https://pith.science/pith/HISRIZNZQDMZIGTIE5EFOEQ3SB","download_json":"https://pith.science/pith/HISRIZNZQDMZIGTIE5EFOEQ3SB.json","view_paper":"https://pith.science/paper/HISRIZNZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.07000&json=true","fetch_graph":"https://pith.science/api/pith-number/HISRIZNZQDMZIGTIE5EFOEQ3SB/graph.json","fetch_events":"https://pith.science/api/pith-number/HISRIZNZQDMZIGTIE5EFOEQ3SB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HISRIZNZQDMZIGTIE5EFOEQ3SB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HISRIZNZQDMZIGTIE5EFOEQ3SB/action/storage_attestation","attest_author":"https://pith.science/pith/HISRIZNZQDMZIGTIE5EFOEQ3SB/action/author_attestation","sign_citation":"https://pith.science/pith/HISRIZNZQDMZIGTIE5EFOEQ3SB/action/citation_signature","submit_replication":"https://pith.science/pith/HISRIZNZQDMZIGTIE5EFOEQ3SB/action/replication_record"}},"created_at":"2026-07-05T09:26:57.462083+00:00","updated_at":"2026-07-05T09:26:57.462083+00:00"}