{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FDOKHPTLWSNGBFQR4Y4I5DYK7A","short_pith_number":"pith:FDOKHPTL","schema_version":"1.0","canonical_sha256":"28dca3be6bb49a609611e6388e8f0af83debd360df968d71b658e6288dcd7fd8","source":{"kind":"arxiv","id":"2305.13788","version":2},"attestation_state":"computed","paper":{"title":"Can Large Language Models Capture Dissenting Human Voices?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"James Thorne, Na Min An, Noah Lee","submitted_at":"2023-05-23T07:55:34Z","abstract_excerpt":"Large language models (LLMs) have shown impressive achievements in solving a broad range of tasks. Augmented by instruction fine-tuning, LLMs have also been shown to generalize in zero-shot settings as well. However, whether LLMs closely align with the human disagreement distribution has not been well-studied, especially within the scope of natural language inference (NLI). In this paper, we evaluate the performance and alignment of LLM distribution with humans using two different techniques to estimate the multinomial distribution: Monte Carlo Estimation (MCE) and Log Probability Estimation ("},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.13788","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-23T07:55:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"dddee534267f8b849d16f702609fb61f01df26a69921ce6c0a1a26b913f42289","abstract_canon_sha256":"ad6d3331d5216cabb0173ff0d919b77e1ee4dd763ac34fae605286d058a86d3c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:43.193544Z","signature_b64":"/JFR/rxArdO4m3DpVXbhNF6/TnUT9+b9Q6f2nkLAB7cyODbCW03WcqI+77gzcP7WeW++cqtWhxbqCS/Cb/gfAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"28dca3be6bb49a609611e6388e8f0af83debd360df968d71b658e6288dcd7fd8","last_reissued_at":"2026-07-05T07:05:43.193065Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:43.193065Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Large Language Models Capture Dissenting Human Voices?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"James Thorne, Na Min An, Noah Lee","submitted_at":"2023-05-23T07:55:34Z","abstract_excerpt":"Large language models (LLMs) have shown impressive achievements in solving a broad range of tasks. Augmented by instruction fine-tuning, LLMs have also been shown to generalize in zero-shot settings as well. However, whether LLMs closely align with the human disagreement distribution has not been well-studied, especially within the scope of natural language inference (NLI). In this paper, we evaluate the performance and alignment of LLM distribution with humans using two different techniques to estimate the multinomial distribution: Monte Carlo Estimation (MCE) and Log Probability Estimation ("},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.13788","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.13788/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.13788","created_at":"2026-07-05T07:05:43.193118+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.13788v2","created_at":"2026-07-05T07:05:43.193118+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.13788","created_at":"2026-07-05T07:05:43.193118+00:00"},{"alias_kind":"pith_short_12","alias_value":"FDOKHPTLWSNG","created_at":"2026-07-05T07:05:43.193118+00:00"},{"alias_kind":"pith_short_16","alias_value":"FDOKHPTLWSNGBFQR","created_at":"2026-07-05T07:05:43.193118+00:00"},{"alias_kind":"pith_short_8","alias_value":"FDOKHPTL","created_at":"2026-07-05T07:05:43.193118+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01168","citing_title":"Quantifying and Predicting Disagreement in Graded Human Ratings","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18069","citing_title":"Modeling Human Perspectives with Socio-Demographic Representations","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FDOKHPTLWSNGBFQR4Y4I5DYK7A","json":"https://pith.science/pith/FDOKHPTLWSNGBFQR4Y4I5DYK7A.json","graph_json":"https://pith.science/api/pith-number/FDOKHPTLWSNGBFQR4Y4I5DYK7A/graph.json","events_json":"https://pith.science/api/pith-number/FDOKHPTLWSNGBFQR4Y4I5DYK7A/events.json","paper":"https://pith.science/paper/FDOKHPTL"},"agent_actions":{"view_html":"https://pith.science/pith/FDOKHPTLWSNGBFQR4Y4I5DYK7A","download_json":"https://pith.science/pith/FDOKHPTLWSNGBFQR4Y4I5DYK7A.json","view_paper":"https://pith.science/paper/FDOKHPTL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.13788&json=true","fetch_graph":"https://pith.science/api/pith-number/FDOKHPTLWSNGBFQR4Y4I5DYK7A/graph.json","fetch_events":"https://pith.science/api/pith-number/FDOKHPTLWSNGBFQR4Y4I5DYK7A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FDOKHPTLWSNGBFQR4Y4I5DYK7A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FDOKHPTLWSNGBFQR4Y4I5DYK7A/action/storage_attestation","attest_author":"https://pith.science/pith/FDOKHPTLWSNGBFQR4Y4I5DYK7A/action/author_attestation","sign_citation":"https://pith.science/pith/FDOKHPTLWSNGBFQR4Y4I5DYK7A/action/citation_signature","submit_replication":"https://pith.science/pith/FDOKHPTLWSNGBFQR4Y4I5DYK7A/action/replication_record"}},"created_at":"2026-07-05T07:05:43.193118+00:00","updated_at":"2026-07-05T07:05:43.193118+00:00"}