{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:TSI5KKW4HMJG57X57D3ZVDUUJY","short_pith_number":"pith:TSI5KKW4","schema_version":"1.0","canonical_sha256":"9c91d52adc3b126efefdf8f79a8e944e1c39b50d2e349aa63cfda4841346dd28","source":{"kind":"arxiv","id":"2211.05853","version":2},"attestation_state":"computed","paper":{"title":"Measuring Reliability of Large Language Models through Semantic Consistency","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Domenic Rosati, Harsh Raj, Subhabrata Majumdar","submitted_at":"2022-11-10T20:21:07Z","abstract_excerpt":"While large pretrained language models (PLMs) demonstrate incredible fluency and performance on many natural language tasks, recent work has shown that well-performing PLMs are very sensitive to what prompts are feed into them. Even when prompts are semantically identical, language models may give very different answers. When considering safe and trustworthy deployments of PLMs we would like their outputs to be consistent under prompts that mean the same thing or convey the same intent. While some work has looked into how state-of-the-art PLMs address this need, they have been limited to only "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.05853","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-11-10T20:21:07Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"a852631445553c0d2b350d96b5ea5efa624e3bf863f7ac5ad5bcac449b60fe5c","abstract_canon_sha256":"b62fa113f162954579a365bd01029b7b4fe9268fdc189ef3cee31506acf7adca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:00:14.384540Z","signature_b64":"JIdU0qxWpHMPB/xzlWnneoqaigObSbPApLxPJXTbCRvsOB+Qx4v2twNWTKu+uI7v/tM7tKgDpeRnWakJPN51DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9c91d52adc3b126efefdf8f79a8e944e1c39b50d2e349aa63cfda4841346dd28","last_reissued_at":"2026-07-05T06:00:14.384079Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:00:14.384079Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measuring Reliability of Large Language Models through Semantic Consistency","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Domenic Rosati, Harsh Raj, Subhabrata Majumdar","submitted_at":"2022-11-10T20:21:07Z","abstract_excerpt":"While large pretrained language models (PLMs) demonstrate incredible fluency and performance on many natural language tasks, recent work has shown that well-performing PLMs are very sensitive to what prompts are feed into them. Even when prompts are semantically identical, language models may give very different answers. When considering safe and trustworthy deployments of PLMs we would like their outputs to be consistent under prompts that mean the same thing or convey the same intent. While some work has looked into how state-of-the-art PLMs address this need, they have been limited to only "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.05853","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.05853/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.05853","created_at":"2026-07-05T06:00:14.384142+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.05853v2","created_at":"2026-07-05T06:00:14.384142+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.05853","created_at":"2026-07-05T06:00:14.384142+00:00"},{"alias_kind":"pith_short_12","alias_value":"TSI5KKW4HMJG","created_at":"2026-07-05T06:00:14.384142+00:00"},{"alias_kind":"pith_short_16","alias_value":"TSI5KKW4HMJG57X5","created_at":"2026-07-05T06:00:14.384142+00:00"},{"alias_kind":"pith_short_8","alias_value":"TSI5KKW4","created_at":"2026-07-05T06:00:14.384142+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.18562","citing_title":"Self-Reported Confidence of Large Language Models in Gastroenterology: Analysis of Commercial, Open-Source, and Quantized Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13875","citing_title":"Common-agency Games for Multi-Objective Test-Time Alignment","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09278","citing_title":"EquiMem: Calibrating Shared Memory in Multi-Agent Debate via Game-Theoretic Equilibrium","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10516","citing_title":"Consistency as a Testable Property: Statistical Methods to Evaluate AI Agent Reliability","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04665","citing_title":"Paraphrase-Induced Output-Mode Collapse: When LLMs Break Character Under Semantically Equivalent Inputs","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04665","citing_title":"Paraphrase-Induced Output-Mode Collapse: When LLMs Break Character Under Semantically Equivalent Inputs","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TSI5KKW4HMJG57X57D3ZVDUUJY","json":"https://pith.science/pith/TSI5KKW4HMJG57X57D3ZVDUUJY.json","graph_json":"https://pith.science/api/pith-number/TSI5KKW4HMJG57X57D3ZVDUUJY/graph.json","events_json":"https://pith.science/api/pith-number/TSI5KKW4HMJG57X57D3ZVDUUJY/events.json","paper":"https://pith.science/paper/TSI5KKW4"},"agent_actions":{"view_html":"https://pith.science/pith/TSI5KKW4HMJG57X57D3ZVDUUJY","download_json":"https://pith.science/pith/TSI5KKW4HMJG57X57D3ZVDUUJY.json","view_paper":"https://pith.science/paper/TSI5KKW4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.05853&json=true","fetch_graph":"https://pith.science/api/pith-number/TSI5KKW4HMJG57X57D3ZVDUUJY/graph.json","fetch_events":"https://pith.science/api/pith-number/TSI5KKW4HMJG57X57D3ZVDUUJY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TSI5KKW4HMJG57X57D3ZVDUUJY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TSI5KKW4HMJG57X57D3ZVDUUJY/action/storage_attestation","attest_author":"https://pith.science/pith/TSI5KKW4HMJG57X57D3ZVDUUJY/action/author_attestation","sign_citation":"https://pith.science/pith/TSI5KKW4HMJG57X57D3ZVDUUJY/action/citation_signature","submit_replication":"https://pith.science/pith/TSI5KKW4HMJG57X57D3ZVDUUJY/action/replication_record"}},"created_at":"2026-07-05T06:00:14.384142+00:00","updated_at":"2026-07-05T06:00:14.384142+00:00"}