{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TDADUTHLZT37JXA7II4S5223F6","short_pith_number":"pith:TDADUTHL","schema_version":"1.0","canonical_sha256":"98c03a4cebccf7f4dc1f42392eeb5b2fa51ae2fc0be2e8e8da48ffab168b82a2","source":{"kind":"arxiv","id":"2407.02996","version":2},"attestation_state":"computed","paper":{"title":"Are Large Language Models Consistent over Value-laden Questions?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Diyi Yang, Jared Moore, Tanvi Deshpande","submitted_at":"2024-07-03T10:53:54Z","abstract_excerpt":"Large language models (LLMs) appear to bias their survey answers toward certain values. Nonetheless, some argue that LLMs are too inconsistent to simulate particular values. Are they? To answer, we first define value consistency as the similarity of answers across (1) paraphrases of one question, (2) related questions under one topic, (3) multiple-choice and open-ended use-cases of one question, and (4) multilingual translations of a question to English, Chinese, German, and Japanese. We apply these measures to small and large, open LLMs including llama-3, as well as gpt-4o, using 8,000 questi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.02996","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-03T10:53:54Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d016f263d1b417292bf4842fe65ff0f113a9efe3170687fc53d0b8c4a4f005fd","abstract_canon_sha256":"5fd442eed94a84fdb8575c921c6443c5a50437fdd72f37827653c36ca3daf1b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:35.350679Z","signature_b64":"pnd0osUXYYDMqJUuwJJLg143GtcRMHHjINUvFVlEOI0PIMQIXw4tgGoGBlfohCl77Dd63TMxtWOHQPeEHshKAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"98c03a4cebccf7f4dc1f42392eeb5b2fa51ae2fc0be2e8e8da48ffab168b82a2","last_reissued_at":"2026-07-05T09:14:35.350114Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:35.350114Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are Large Language Models Consistent over Value-laden Questions?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Diyi Yang, Jared Moore, Tanvi Deshpande","submitted_at":"2024-07-03T10:53:54Z","abstract_excerpt":"Large language models (LLMs) appear to bias their survey answers toward certain values. Nonetheless, some argue that LLMs are too inconsistent to simulate particular values. Are they? To answer, we first define value consistency as the similarity of answers across (1) paraphrases of one question, (2) related questions under one topic, (3) multiple-choice and open-ended use-cases of one question, and (4) multilingual translations of a question to English, Chinese, German, and Japanese. We apply these measures to small and large, open LLMs including llama-3, as well as gpt-4o, using 8,000 questi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.02996","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.02996/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.02996","created_at":"2026-07-05T09:14:35.350172+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.02996v2","created_at":"2026-07-05T09:14:35.350172+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.02996","created_at":"2026-07-05T09:14:35.350172+00:00"},{"alias_kind":"pith_short_12","alias_value":"TDADUTHLZT37","created_at":"2026-07-05T09:14:35.350172+00:00"},{"alias_kind":"pith_short_16","alias_value":"TDADUTHLZT37JXA7","created_at":"2026-07-05T09:14:35.350172+00:00"},{"alias_kind":"pith_short_8","alias_value":"TDADUTHL","created_at":"2026-07-05T09:14:35.350172+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12731","citing_title":"Normative Robustness as a Frontier for Non-Verifiable Reasoning in LLMs","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30036","citing_title":"Teaching Values to Machines: Simulating Human-Like Behavior in LLMs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05330","citing_title":"A Model of Multi-turn Human Persuadability Using Probabilistic Belief Tracing","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2502.19463","citing_title":"Hedging and Non-Affirmation: Quantifying LLM Alignment on Questions of Human Rights","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20512","citing_title":"Framing an AI with Values Reduces AI Reliance in AI-supported Writing Tasks","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03255","citing_title":"Do LLMs have core beliefs?","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TDADUTHLZT37JXA7II4S5223F6","json":"https://pith.science/pith/TDADUTHLZT37JXA7II4S5223F6.json","graph_json":"https://pith.science/api/pith-number/TDADUTHLZT37JXA7II4S5223F6/graph.json","events_json":"https://pith.science/api/pith-number/TDADUTHLZT37JXA7II4S5223F6/events.json","paper":"https://pith.science/paper/TDADUTHL"},"agent_actions":{"view_html":"https://pith.science/pith/TDADUTHLZT37JXA7II4S5223F6","download_json":"https://pith.science/pith/TDADUTHLZT37JXA7II4S5223F6.json","view_paper":"https://pith.science/paper/TDADUTHL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.02996&json=true","fetch_graph":"https://pith.science/api/pith-number/TDADUTHLZT37JXA7II4S5223F6/graph.json","fetch_events":"https://pith.science/api/pith-number/TDADUTHLZT37JXA7II4S5223F6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TDADUTHLZT37JXA7II4S5223F6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TDADUTHLZT37JXA7II4S5223F6/action/storage_attestation","attest_author":"https://pith.science/pith/TDADUTHLZT37JXA7II4S5223F6/action/author_attestation","sign_citation":"https://pith.science/pith/TDADUTHLZT37JXA7II4S5223F6/action/citation_signature","submit_replication":"https://pith.science/pith/TDADUTHLZT37JXA7II4S5223F6/action/replication_record"}},"created_at":"2026-07-05T09:14:35.350172+00:00","updated_at":"2026-07-05T09:14:35.350172+00:00"}