{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CH5J7WTQ3O7VPF5DSK6ZXOGDXL","short_pith_number":"pith:CH5J7WTQ","schema_version":"1.0","canonical_sha256":"11fa9fda70dbbf5797a392bd9bb8c3baedcb78b828c3db9ecb337bf6facba534","source":{"kind":"arxiv","id":"2311.07383","version":1},"attestation_state":"computed","paper":{"title":"LM-Polygraph: Uncertainty Estimation for Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Akim Tsvigun, Alexander Panchenko, Artem Shelmanov, Artem Vazhentsev, Daniil Vasilev, Ekaterina Fadeeva, Elizaveta Goncharova, Kirill Fedyanin, Maxim Panov, Roman Vashurin, Sergey Petrakov, Timothy Baldwin","submitted_at":"2023-11-13T15:08:59Z","abstract_excerpt":"Recent advancements in the capabilities of large language models (LLMs) have paved the way for a myriad of groundbreaking applications in various fields. However, a significant challenge arises as these models often \"hallucinate\", i.e., fabricate facts without providing users an apparent means to discern the veracity of their statements. Uncertainty estimation (UE) methods are one path to safer, more responsible, and more effective use of LLMs. However, to date, research on UE methods for LLMs has been focused primarily on theoretical rather than engineering contributions. In this work, we tac"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.07383","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-13T15:08:59Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e1c386cc86cd264a40ac85b97746bfdaca7ce4517018ce384f7679adc2900e0d","abstract_canon_sha256":"5a56f825fa65887abdbe2d006d3bcc5567a7bd9c7b0a9821005c7e6ead5602f0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:11.904019Z","signature_b64":"2rTamg31LTQLsPDXtZlqjQxB7EnGQ/BLOnTqL0PhX5F38snJeAUAYMAGevfNSIUMg0jyKnf4fnBUnRSipjm6Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"11fa9fda70dbbf5797a392bd9bb8c3baedcb78b828c3db9ecb337bf6facba534","last_reissued_at":"2026-07-05T07:12:11.903589Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:11.903589Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LM-Polygraph: Uncertainty Estimation for Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Akim Tsvigun, Alexander Panchenko, Artem Shelmanov, Artem Vazhentsev, Daniil Vasilev, Ekaterina Fadeeva, Elizaveta Goncharova, Kirill Fedyanin, Maxim Panov, Roman Vashurin, Sergey Petrakov, Timothy Baldwin","submitted_at":"2023-11-13T15:08:59Z","abstract_excerpt":"Recent advancements in the capabilities of large language models (LLMs) have paved the way for a myriad of groundbreaking applications in various fields. However, a significant challenge arises as these models often \"hallucinate\", i.e., fabricate facts without providing users an apparent means to discern the veracity of their statements. Uncertainty estimation (UE) methods are one path to safer, more responsible, and more effective use of LLMs. However, to date, research on UE methods for LLMs has been focused primarily on theoretical rather than engineering contributions. In this work, we tac"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.07383","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.07383/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.07383","created_at":"2026-07-05T07:12:11.903646+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.07383v1","created_at":"2026-07-05T07:12:11.903646+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.07383","created_at":"2026-07-05T07:12:11.903646+00:00"},{"alias_kind":"pith_short_12","alias_value":"CH5J7WTQ3O7V","created_at":"2026-07-05T07:12:11.903646+00:00"},{"alias_kind":"pith_short_16","alias_value":"CH5J7WTQ3O7VPF5D","created_at":"2026-07-05T07:12:11.903646+00:00"},{"alias_kind":"pith_short_8","alias_value":"CH5J7WTQ","created_at":"2026-07-05T07:12:11.903646+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25760","citing_title":"Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2502.14427","citing_title":"Token-Level Density-Based Uncertainty Quantification Methods for Eliciting Truthfulness of Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05348","citing_title":"From Retinal Evidence to Safe Decisions: RETINA-SAFE and ECRT for Hallucination Risk Triage in Medical LLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17112","citing_title":"Complementing Self-Consistency with Cross-Model Disagreement for Uncertainty Quantification","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CH5J7WTQ3O7VPF5DSK6ZXOGDXL","json":"https://pith.science/pith/CH5J7WTQ3O7VPF5DSK6ZXOGDXL.json","graph_json":"https://pith.science/api/pith-number/CH5J7WTQ3O7VPF5DSK6ZXOGDXL/graph.json","events_json":"https://pith.science/api/pith-number/CH5J7WTQ3O7VPF5DSK6ZXOGDXL/events.json","paper":"https://pith.science/paper/CH5J7WTQ"},"agent_actions":{"view_html":"https://pith.science/pith/CH5J7WTQ3O7VPF5DSK6ZXOGDXL","download_json":"https://pith.science/pith/CH5J7WTQ3O7VPF5DSK6ZXOGDXL.json","view_paper":"https://pith.science/paper/CH5J7WTQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.07383&json=true","fetch_graph":"https://pith.science/api/pith-number/CH5J7WTQ3O7VPF5DSK6ZXOGDXL/graph.json","fetch_events":"https://pith.science/api/pith-number/CH5J7WTQ3O7VPF5DSK6ZXOGDXL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CH5J7WTQ3O7VPF5DSK6ZXOGDXL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CH5J7WTQ3O7VPF5DSK6ZXOGDXL/action/storage_attestation","attest_author":"https://pith.science/pith/CH5J7WTQ3O7VPF5DSK6ZXOGDXL/action/author_attestation","sign_citation":"https://pith.science/pith/CH5J7WTQ3O7VPF5DSK6ZXOGDXL/action/citation_signature","submit_replication":"https://pith.science/pith/CH5J7WTQ3O7VPF5DSK6ZXOGDXL/action/replication_record"}},"created_at":"2026-07-05T07:12:11.903646+00:00","updated_at":"2026-07-05T07:12:11.903646+00:00"}