{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HRUYW62KRE2RH5UGLKMC7I53E5","short_pith_number":"pith:HRUYW62K","schema_version":"1.0","canonical_sha256":"3c698b7b4a893513f6865a982fa3bb27492e68ebbec465f9afda886dfb73c2c2","source":{"kind":"arxiv","id":"2502.11677","version":2},"attestation_state":"computed","paper":{"title":"Towards Fully Exploiting LLM Internal States to Enhance Knowledge Boundary Perception","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baolong Bi, Jiafeng Guo, Keping Bi, Lulu Yu, Shiyu Ni, Xueqi Cheng","submitted_at":"2025-02-17T11:11:09Z","abstract_excerpt":"Large language models (LLMs) exhibit impressive performance across diverse tasks but often struggle to accurately gauge their knowledge boundaries, leading to confident yet incorrect responses. This paper explores leveraging LLMs' internal states to enhance their perception of knowledge boundaries from efficiency and risk perspectives. We investigate whether LLMs can estimate their confidence using internal states before response generation, potentially saving computational resources. Our experiments on datasets like Natural Questions, HotpotQA, and MMLU reveal that LLMs demonstrate significan"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.11677","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-17T11:11:09Z","cross_cats_sorted":[],"title_canon_sha256":"378771d75fb6a16b24781e5cc873d47055da98eee80e04c910e956cc634e5870","abstract_canon_sha256":"3402f089969a892f7a1fa20421e2818f005c4507d4659131cf02906122384a90"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:26:47.852728Z","signature_b64":"ZWyic2vikKNj3aviXUAQ+mHev7GZ/RIxgaAZM9f8f9YTFUdhhj+tueXe3tS/gq/yL9Ky0SJ5VSGUjTxhPv1rAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c698b7b4a893513f6865a982fa3bb27492e68ebbec465f9afda886dfb73c2c2","last_reissued_at":"2026-07-05T11:26:47.852241Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:26:47.852241Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Fully Exploiting LLM Internal States to Enhance Knowledge Boundary Perception","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baolong Bi, Jiafeng Guo, Keping Bi, Lulu Yu, Shiyu Ni, Xueqi Cheng","submitted_at":"2025-02-17T11:11:09Z","abstract_excerpt":"Large language models (LLMs) exhibit impressive performance across diverse tasks but often struggle to accurately gauge their knowledge boundaries, leading to confident yet incorrect responses. This paper explores leveraging LLMs' internal states to enhance their perception of knowledge boundaries from efficiency and risk perspectives. We investigate whether LLMs can estimate their confidence using internal states before response generation, potentially saving computational resources. Our experiments on datasets like Natural Questions, HotpotQA, and MMLU reveal that LLMs demonstrate significan"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.11677","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.11677/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.11677","created_at":"2026-07-05T11:26:47.852296+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.11677v2","created_at":"2026-07-05T11:26:47.852296+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.11677","created_at":"2026-07-05T11:26:47.852296+00:00"},{"alias_kind":"pith_short_12","alias_value":"HRUYW62KRE2R","created_at":"2026-07-05T11:26:47.852296+00:00"},{"alias_kind":"pith_short_16","alias_value":"HRUYW62KRE2RH5UG","created_at":"2026-07-05T11:26:47.852296+00:00"},{"alias_kind":"pith_short_8","alias_value":"HRUYW62K","created_at":"2026-07-05T11:26:47.852296+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03535","citing_title":"Can LLM Rerankers Predict Their Own Ranking Performance?","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09117","citing_title":"Decoupling Reasoning and Confidence: Resurrecting Calibration in Reinforcement Learning from Verifiable Rewards","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HRUYW62KRE2RH5UGLKMC7I53E5","json":"https://pith.science/pith/HRUYW62KRE2RH5UGLKMC7I53E5.json","graph_json":"https://pith.science/api/pith-number/HRUYW62KRE2RH5UGLKMC7I53E5/graph.json","events_json":"https://pith.science/api/pith-number/HRUYW62KRE2RH5UGLKMC7I53E5/events.json","paper":"https://pith.science/paper/HRUYW62K"},"agent_actions":{"view_html":"https://pith.science/pith/HRUYW62KRE2RH5UGLKMC7I53E5","download_json":"https://pith.science/pith/HRUYW62KRE2RH5UGLKMC7I53E5.json","view_paper":"https://pith.science/paper/HRUYW62K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.11677&json=true","fetch_graph":"https://pith.science/api/pith-number/HRUYW62KRE2RH5UGLKMC7I53E5/graph.json","fetch_events":"https://pith.science/api/pith-number/HRUYW62KRE2RH5UGLKMC7I53E5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HRUYW62KRE2RH5UGLKMC7I53E5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HRUYW62KRE2RH5UGLKMC7I53E5/action/storage_attestation","attest_author":"https://pith.science/pith/HRUYW62KRE2RH5UGLKMC7I53E5/action/author_attestation","sign_citation":"https://pith.science/pith/HRUYW62KRE2RH5UGLKMC7I53E5/action/citation_signature","submit_replication":"https://pith.science/pith/HRUYW62KRE2RH5UGLKMC7I53E5/action/replication_record"}},"created_at":"2026-07-05T11:26:47.852296+00:00","updated_at":"2026-07-05T11:26:47.852296+00:00"}