{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7E3ZIBNC25KUFCEFTXRIPOTQYB","short_pith_number":"pith:7E3ZIBNC","schema_version":"1.0","canonical_sha256":"f9379405a2d7554288859de287ba70c0401eec51a9bb909e76743086e677e7d9","source":{"kind":"arxiv","id":"2509.04464","version":1},"attestation_state":"computed","paper":{"title":"Can Multiple Responses from an LLM Reveal the Sources of Its Uncertainty?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Han Xu, Pengfei He, Ravi Tandon, Yang Nan","submitted_at":"2025-08-28T20:14:35Z","abstract_excerpt":"Large language models (LLMs) have delivered significant breakthroughs across diverse domains but can still produce unreliable or misleading outputs, posing critical challenges for real-world applications. While many recent studies focus on quantifying model uncertainty, relatively little work has been devoted to \\textit{diagnosing the source of uncertainty}. In this study, we show that, when an LLM is uncertain, the patterns of disagreement among its multiple generated responses contain rich clues about the underlying cause of uncertainty. To illustrate this point, we collect multiple response"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.04464","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-08-28T20:14:35Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e4ab835865189e56f2d982f206f67d7727ff4f6c9bf5b985ac504de798ae984d","abstract_canon_sha256":"6923599dc9c13e02b6a59e1c4ed99de4ad585ed4b09bdbb339bb56c4cd7fb630"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:05:23.493353Z","signature_b64":"NSi+ZeeFmxHKO17nQ06Acgh57p9VEgKxHg5nsKMyWquKigHlnyIc7m6P5ucVN1DldGFzPIASYchD4eJ6K30vAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f9379405a2d7554288859de287ba70c0401eec51a9bb909e76743086e677e7d9","last_reissued_at":"2026-07-05T12:05:23.492842Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:05:23.492842Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Multiple Responses from an LLM Reveal the Sources of Its Uncertainty?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Han Xu, Pengfei He, Ravi Tandon, Yang Nan","submitted_at":"2025-08-28T20:14:35Z","abstract_excerpt":"Large language models (LLMs) have delivered significant breakthroughs across diverse domains but can still produce unreliable or misleading outputs, posing critical challenges for real-world applications. While many recent studies focus on quantifying model uncertainty, relatively little work has been devoted to \\textit{diagnosing the source of uncertainty}. In this study, we show that, when an LLM is uncertain, the patterns of disagreement among its multiple generated responses contain rich clues about the underlying cause of uncertainty. To illustrate this point, we collect multiple response"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.04464","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.04464/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.04464","created_at":"2026-07-05T12:05:23.492903+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.04464v1","created_at":"2026-07-05T12:05:23.492903+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.04464","created_at":"2026-07-05T12:05:23.492903+00:00"},{"alias_kind":"pith_short_12","alias_value":"7E3ZIBNC25KU","created_at":"2026-07-05T12:05:23.492903+00:00"},{"alias_kind":"pith_short_16","alias_value":"7E3ZIBNC25KUFCEF","created_at":"2026-07-05T12:05:23.492903+00:00"},{"alias_kind":"pith_short_8","alias_value":"7E3ZIBNC","created_at":"2026-07-05T12:05:23.492903+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.20012","citing_title":"Reliable LLM-Based Edge-Cloud-Expert Cascades for Telecom Knowledge Systems","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7E3ZIBNC25KUFCEFTXRIPOTQYB","json":"https://pith.science/pith/7E3ZIBNC25KUFCEFTXRIPOTQYB.json","graph_json":"https://pith.science/api/pith-number/7E3ZIBNC25KUFCEFTXRIPOTQYB/graph.json","events_json":"https://pith.science/api/pith-number/7E3ZIBNC25KUFCEFTXRIPOTQYB/events.json","paper":"https://pith.science/paper/7E3ZIBNC"},"agent_actions":{"view_html":"https://pith.science/pith/7E3ZIBNC25KUFCEFTXRIPOTQYB","download_json":"https://pith.science/pith/7E3ZIBNC25KUFCEFTXRIPOTQYB.json","view_paper":"https://pith.science/paper/7E3ZIBNC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.04464&json=true","fetch_graph":"https://pith.science/api/pith-number/7E3ZIBNC25KUFCEFTXRIPOTQYB/graph.json","fetch_events":"https://pith.science/api/pith-number/7E3ZIBNC25KUFCEFTXRIPOTQYB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7E3ZIBNC25KUFCEFTXRIPOTQYB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7E3ZIBNC25KUFCEFTXRIPOTQYB/action/storage_attestation","attest_author":"https://pith.science/pith/7E3ZIBNC25KUFCEFTXRIPOTQYB/action/author_attestation","sign_citation":"https://pith.science/pith/7E3ZIBNC25KUFCEFTXRIPOTQYB/action/citation_signature","submit_replication":"https://pith.science/pith/7E3ZIBNC25KUFCEFTXRIPOTQYB/action/replication_record"}},"created_at":"2026-07-05T12:05:23.492903+00:00","updated_at":"2026-07-05T12:05:23.492903+00:00"}