{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:BULA2RWJF4YXGYJF4RASB3D4RP","short_pith_number":"pith:BULA2RWJ","schema_version":"1.0","canonical_sha256":"0d160d46c92f31736125e44120ec7c8bf812fba9ebfdda05fb0084a09ab3089a","source":{"kind":"arxiv","id":"2607.11414","version":1},"attestation_state":"computed","paper":{"title":"Confidently Wrong: Detecting Hallucinations in Financial Question Answering from LLM Internal States","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Richard Zhe Wang","submitted_at":"2026-07-13T11:22:25Z","abstract_excerpt":"Large language models (LLMs) in financial applications fail most consequentially when they are confidently wrong. Hedged, uncertain answers invite scrutiny, whereas confident errors silently degrade downstream decisions without warning. We ask how reliably such confidently wrong answers, or confident hallucinations, can be detected from a model's internal activations, and whether those activations carry information beyond its observable outputs. We train linear probes on the residual stream and evaluate them on two established question-answering (QA) benchmarks built from real filings, FinQA a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.11414","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-07-13T11:22:25Z","cross_cats_sorted":[],"title_canon_sha256":"46a5dcd4f9a9548161d653f8d530c5ea6c0847c5bf64a6ae7652bfc6a4da9640","abstract_canon_sha256":"bc72fd351e9c89b5ae3288a68e663efa626a0d28a2182aea614d6f5a846def32"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T02:22:00.973289Z","signature_b64":"PUuEUux4J5SPEeki/9zEhhQp33LDuxqFBfvsZPE5SCnrisNd76m/Hwn7JnWr00icgu58bUDGJEk2QQHTQW7tDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0d160d46c92f31736125e44120ec7c8bf812fba9ebfdda05fb0084a09ab3089a","last_reissued_at":"2026-07-14T02:22:00.972534Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T02:22:00.972534Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Confidently Wrong: Detecting Hallucinations in Financial Question Answering from LLM Internal States","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Richard Zhe Wang","submitted_at":"2026-07-13T11:22:25Z","abstract_excerpt":"Large language models (LLMs) in financial applications fail most consequentially when they are confidently wrong. Hedged, uncertain answers invite scrutiny, whereas confident errors silently degrade downstream decisions without warning. We ask how reliably such confidently wrong answers, or confident hallucinations, can be detected from a model's internal activations, and whether those activations carry information beyond its observable outputs. We train linear probes on the residual stream and evaluate them on two established question-answering (QA) benchmarks built from real filings, FinQA a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.11414","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.11414/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.11414","created_at":"2026-07-14T02:22:00.972943+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.11414v1","created_at":"2026-07-14T02:22:00.972943+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.11414","created_at":"2026-07-14T02:22:00.972943+00:00"},{"alias_kind":"pith_short_12","alias_value":"BULA2RWJF4YX","created_at":"2026-07-14T02:22:00.972943+00:00"},{"alias_kind":"pith_short_16","alias_value":"BULA2RWJF4YXGYJF","created_at":"2026-07-14T02:22:00.972943+00:00"},{"alias_kind":"pith_short_8","alias_value":"BULA2RWJ","created_at":"2026-07-14T02:22:00.972943+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BULA2RWJF4YXGYJF4RASB3D4RP","json":"https://pith.science/pith/BULA2RWJF4YXGYJF4RASB3D4RP.json","graph_json":"https://pith.science/api/pith-number/BULA2RWJF4YXGYJF4RASB3D4RP/graph.json","events_json":"https://pith.science/api/pith-number/BULA2RWJF4YXGYJF4RASB3D4RP/events.json","paper":"https://pith.science/paper/BULA2RWJ"},"agent_actions":{"view_html":"https://pith.science/pith/BULA2RWJF4YXGYJF4RASB3D4RP","download_json":"https://pith.science/pith/BULA2RWJF4YXGYJF4RASB3D4RP.json","view_paper":"https://pith.science/paper/BULA2RWJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.11414&json=true","fetch_graph":"https://pith.science/api/pith-number/BULA2RWJF4YXGYJF4RASB3D4RP/graph.json","fetch_events":"https://pith.science/api/pith-number/BULA2RWJF4YXGYJF4RASB3D4RP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BULA2RWJF4YXGYJF4RASB3D4RP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BULA2RWJF4YXGYJF4RASB3D4RP/action/storage_attestation","attest_author":"https://pith.science/pith/BULA2RWJF4YXGYJF4RASB3D4RP/action/author_attestation","sign_citation":"https://pith.science/pith/BULA2RWJF4YXGYJF4RASB3D4RP/action/citation_signature","submit_replication":"https://pith.science/pith/BULA2RWJF4YXGYJF4RASB3D4RP/action/replication_record"}},"created_at":"2026-07-14T02:22:00.972943+00:00","updated_at":"2026-07-14T02:22:00.972943+00:00"}