{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LLZZVVLJJQCV7PNBXBRWVCSKLF","short_pith_number":"pith:LLZZVVLJ","schema_version":"1.0","canonical_sha256":"5af39ad5694c055fbda1b8636a8a4a594e85889aa5340803c41e1f92ebdae2f1","source":{"kind":"arxiv","id":"2402.00367","version":2},"attestation_state":"computed","paper":{"title":"Don't Hallucinate, Abstain: Identifying LLM Knowledge Gaps via Multi-LLM Collaboration","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Shangbin Feng, Vidhisha Balachandran, Weijia Shi, Wenxuan Ding, Yike Wang, Yulia Tsvetkov","submitted_at":"2024-02-01T06:11:49Z","abstract_excerpt":"Despite efforts to expand the knowledge of large language models (LLMs), knowledge gaps -- missing or outdated information in LLMs -- might always persist given the evolving nature of knowledge. In this work, we study approaches to identify LLM knowledge gaps and abstain from answering questions when knowledge gaps are present. We first adapt existing approaches to model calibration or adaptation through fine-tuning/prompting and analyze their ability to abstain from generating low-confidence outputs. Motivated by their failures in self-reflection and over-reliance on held-out sets, we propose"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.00367","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-01T06:11:49Z","cross_cats_sorted":[],"title_canon_sha256":"c4906f5d68117c3ec87dd03724035d9a680a0319ca2d8d6d929dd157b52e2508","abstract_canon_sha256":"fe95acf333eec86cdc7f52ce401402c4982b7a2c0a0332dd9204eaf3285c9e80"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:05.693059Z","signature_b64":"wj1I7oz7YTm6WqWWduoYfNrPvBK4FQ33qStEH4AJjE3LcIklNdc9V59Ro9uEFP+/5ecpDsfp94GXd/EGoOARAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5af39ad5694c055fbda1b8636a8a4a594e85889aa5340803c41e1f92ebdae2f1","last_reissued_at":"2026-07-05T08:38:05.692537Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:05.692537Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Don't Hallucinate, Abstain: Identifying LLM Knowledge Gaps via Multi-LLM Collaboration","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Shangbin Feng, Vidhisha Balachandran, Weijia Shi, Wenxuan Ding, Yike Wang, Yulia Tsvetkov","submitted_at":"2024-02-01T06:11:49Z","abstract_excerpt":"Despite efforts to expand the knowledge of large language models (LLMs), knowledge gaps -- missing or outdated information in LLMs -- might always persist given the evolving nature of knowledge. In this work, we study approaches to identify LLM knowledge gaps and abstain from answering questions when knowledge gaps are present. We first adapt existing approaches to model calibration or adaptation through fine-tuning/prompting and analyze their ability to abstain from generating low-confidence outputs. Motivated by their failures in self-reflection and over-reliance on held-out sets, we propose"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.00367","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.00367/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.00367","created_at":"2026-07-05T08:38:05.692606+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.00367v2","created_at":"2026-07-05T08:38:05.692606+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.00367","created_at":"2026-07-05T08:38:05.692606+00:00"},{"alias_kind":"pith_short_12","alias_value":"LLZZVVLJJQCV","created_at":"2026-07-05T08:38:05.692606+00:00"},{"alias_kind":"pith_short_16","alias_value":"LLZZVVLJJQCV7PNB","created_at":"2026-07-05T08:38:05.692606+00:00"},{"alias_kind":"pith_short_8","alias_value":"LLZZVVLJ","created_at":"2026-07-05T08:38:05.692606+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24370","citing_title":"When Helpfulness Overrides Causal Caution: Context-Dependent Suppression and Recovery in LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31010","citing_title":"MoG: Mixture of Experts for Graph-based Retrieval-Augmented Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01188","citing_title":"ZoFia: Zero-Shot Fake News Detection with Entity-Guided Retrieval and Multi-LLM Interaction","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07769","citing_title":"Coding Agents Don't Know When to Act","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04565","citing_title":"PassiveQA: A Three-Action Framework for Epistemically Calibrated Question Answering via Supervised Finetuning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17200","citing_title":"Calibrating Model-Based Evaluation Metrics for Summarization","ref_index":103,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LLZZVVLJJQCV7PNBXBRWVCSKLF","json":"https://pith.science/pith/LLZZVVLJJQCV7PNBXBRWVCSKLF.json","graph_json":"https://pith.science/api/pith-number/LLZZVVLJJQCV7PNBXBRWVCSKLF/graph.json","events_json":"https://pith.science/api/pith-number/LLZZVVLJJQCV7PNBXBRWVCSKLF/events.json","paper":"https://pith.science/paper/LLZZVVLJ"},"agent_actions":{"view_html":"https://pith.science/pith/LLZZVVLJJQCV7PNBXBRWVCSKLF","download_json":"https://pith.science/pith/LLZZVVLJJQCV7PNBXBRWVCSKLF.json","view_paper":"https://pith.science/paper/LLZZVVLJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.00367&json=true","fetch_graph":"https://pith.science/api/pith-number/LLZZVVLJJQCV7PNBXBRWVCSKLF/graph.json","fetch_events":"https://pith.science/api/pith-number/LLZZVVLJJQCV7PNBXBRWVCSKLF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LLZZVVLJJQCV7PNBXBRWVCSKLF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LLZZVVLJJQCV7PNBXBRWVCSKLF/action/storage_attestation","attest_author":"https://pith.science/pith/LLZZVVLJJQCV7PNBXBRWVCSKLF/action/author_attestation","sign_citation":"https://pith.science/pith/LLZZVVLJJQCV7PNBXBRWVCSKLF/action/citation_signature","submit_replication":"https://pith.science/pith/LLZZVVLJJQCV7PNBXBRWVCSKLF/action/replication_record"}},"created_at":"2026-07-05T08:38:05.692606+00:00","updated_at":"2026-07-05T08:38:05.692606+00:00"}