{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:75HNDYYECZVUU36NIAUYC6ILB6","short_pith_number":"pith:75HNDYYE","schema_version":"1.0","canonical_sha256":"ff4ed1e304166b4a6fcd402981790b0fbc6668b9025a8589c0a878f9ff4f941e","source":{"kind":"arxiv","id":"2308.09138","version":2},"attestation_state":"computed","paper":{"title":"Semantic Consistency for Assuring Reliability of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Domenic Rosati, Harsh Raj, Subhabrata Majumdar, Vipul Gupta","submitted_at":"2023-08-17T18:11:33Z","abstract_excerpt":"Large Language Models (LLMs) exhibit remarkable fluency and competence across various natural language tasks. However, recent research has highlighted their sensitivity to variations in input prompts. To deploy LLMs in a safe and reliable manner, it is crucial for their outputs to be consistent when prompted with expressions that carry the same meaning or intent. While some existing work has explored how state-of-the-art LLMs address this issue, their evaluations have been confined to assessing lexical equality of single- or multi-word answers, overlooking the consistency of generative text se"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.09138","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-17T18:11:33Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"54efd1c772482c593a30333427ea15964eb55f115d7f0e7cc9381cf1c550dcad","abstract_canon_sha256":"300cad026b7b4bda6ee85a59cc8600f17de5263d6916ae498f933a967fac3583"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:55:16.791496Z","signature_b64":"mu2F301zwSQrwk035+Ok2P/WItWr7xljePtK8zYx71A0agDoeQJvAAacTcDiJQH3IWJEE2aW2z+dSkhYSSeuCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff4ed1e304166b4a6fcd402981790b0fbc6668b9025a8589c0a878f9ff4f941e","last_reissued_at":"2026-07-05T10:55:16.790962Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:55:16.790962Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Semantic Consistency for Assuring Reliability of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Domenic Rosati, Harsh Raj, Subhabrata Majumdar, Vipul Gupta","submitted_at":"2023-08-17T18:11:33Z","abstract_excerpt":"Large Language Models (LLMs) exhibit remarkable fluency and competence across various natural language tasks. However, recent research has highlighted their sensitivity to variations in input prompts. To deploy LLMs in a safe and reliable manner, it is crucial for their outputs to be consistent when prompted with expressions that carry the same meaning or intent. While some existing work has explored how state-of-the-art LLMs address this issue, their evaluations have been confined to assessing lexical equality of single- or multi-word answers, overlooking the consistency of generative text se"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.09138","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.09138/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.09138","created_at":"2026-07-05T10:55:16.791016+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.09138v2","created_at":"2026-07-05T10:55:16.791016+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.09138","created_at":"2026-07-05T10:55:16.791016+00:00"},{"alias_kind":"pith_short_12","alias_value":"75HNDYYECZVU","created_at":"2026-07-05T10:55:16.791016+00:00"},{"alias_kind":"pith_short_16","alias_value":"75HNDYYECZVUU36N","created_at":"2026-07-05T10:55:16.791016+00:00"},{"alias_kind":"pith_short_8","alias_value":"75HNDYYE","created_at":"2026-07-05T10:55:16.791016+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.19513","citing_title":"Visual hallucination detection in large vision-language models via evidential conflict","ref_index":65,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/75HNDYYECZVUU36NIAUYC6ILB6","json":"https://pith.science/pith/75HNDYYECZVUU36NIAUYC6ILB6.json","graph_json":"https://pith.science/api/pith-number/75HNDYYECZVUU36NIAUYC6ILB6/graph.json","events_json":"https://pith.science/api/pith-number/75HNDYYECZVUU36NIAUYC6ILB6/events.json","paper":"https://pith.science/paper/75HNDYYE"},"agent_actions":{"view_html":"https://pith.science/pith/75HNDYYECZVUU36NIAUYC6ILB6","download_json":"https://pith.science/pith/75HNDYYECZVUU36NIAUYC6ILB6.json","view_paper":"https://pith.science/paper/75HNDYYE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.09138&json=true","fetch_graph":"https://pith.science/api/pith-number/75HNDYYECZVUU36NIAUYC6ILB6/graph.json","fetch_events":"https://pith.science/api/pith-number/75HNDYYECZVUU36NIAUYC6ILB6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/75HNDYYECZVUU36NIAUYC6ILB6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/75HNDYYECZVUU36NIAUYC6ILB6/action/storage_attestation","attest_author":"https://pith.science/pith/75HNDYYECZVUU36NIAUYC6ILB6/action/author_attestation","sign_citation":"https://pith.science/pith/75HNDYYECZVUU36NIAUYC6ILB6/action/citation_signature","submit_replication":"https://pith.science/pith/75HNDYYECZVUU36NIAUYC6ILB6/action/replication_record"}},"created_at":"2026-07-05T10:55:16.791016+00:00","updated_at":"2026-07-05T10:55:16.791016+00:00"}