{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:D4YRKDJZLKQVSQUYAXTJA2V3YT","short_pith_number":"pith:D4YRKDJZ","schema_version":"1.0","canonical_sha256":"1f31150d395aa159429805e6906abbc4ebd22519cd7f0a5fdbeb8631b9b211f0","source":{"kind":"arxiv","id":"2503.08688","version":2},"attestation_state":"computed","paper":{"title":"Randomness, Not Representation: The Unreliability of Evaluating Cultural Alignment in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CY","authors_text":"Ariba Khan, Dylan Hadfield-Menell, Stephen Casper","submitted_at":"2025-03-11T17:59:53Z","abstract_excerpt":"Research on the 'cultural alignment' of Large Language Models (LLMs) has emerged in response to growing interest in understanding representation across diverse stakeholders. Current approaches to evaluating cultural alignment through survey-based assessments that borrow from social science methodologies often overlook systematic robustness checks. Here, we identify and test three assumptions behind current survey-based evaluation methods: (1) Stability: that cultural alignment is a property of LLMs rather than an artifact of evaluation design, (2) Extrapolability: that alignment with one cultu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.08688","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2025-03-11T17:59:53Z","cross_cats_sorted":[],"title_canon_sha256":"6807dc66d46995022648534d43d7311cb1ba963d2ed7f0707f14a7edbf86d817","abstract_canon_sha256":"0449277cf47460bb5b1ead6cad71e5091961c65a37f59f515bc0137c837b39d3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:46:23.094368Z","signature_b64":"KrocLoKW5pY0JcgLFdWjqjH3D4qOGp2KElmZYmSx6qvF9iBwqXT7p+wY8J+pRAi+79jUp06u0Aj5PVhL4xJADA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f31150d395aa159429805e6906abbc4ebd22519cd7f0a5fdbeb8631b9b211f0","last_reissued_at":"2026-07-05T10:46:23.093784Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:46:23.093784Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Randomness, Not Representation: The Unreliability of Evaluating Cultural Alignment in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CY","authors_text":"Ariba Khan, Dylan Hadfield-Menell, Stephen Casper","submitted_at":"2025-03-11T17:59:53Z","abstract_excerpt":"Research on the 'cultural alignment' of Large Language Models (LLMs) has emerged in response to growing interest in understanding representation across diverse stakeholders. Current approaches to evaluating cultural alignment through survey-based assessments that borrow from social science methodologies often overlook systematic robustness checks. Here, we identify and test three assumptions behind current survey-based evaluation methods: (1) Stability: that cultural alignment is a property of LLMs rather than an artifact of evaluation design, (2) Extrapolability: that alignment with one cultu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.08688","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.08688/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.08688","created_at":"2026-07-05T10:46:23.093846+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.08688v2","created_at":"2026-07-05T10:46:23.093846+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.08688","created_at":"2026-07-05T10:46:23.093846+00:00"},{"alias_kind":"pith_short_12","alias_value":"D4YRKDJZLKQV","created_at":"2026-07-05T10:46:23.093846+00:00"},{"alias_kind":"pith_short_16","alias_value":"D4YRKDJZLKQVSQUY","created_at":"2026-07-05T10:46:23.093846+00:00"},{"alias_kind":"pith_short_8","alias_value":"D4YRKDJZ","created_at":"2026-07-05T10:46:23.093846+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12730","citing_title":"Rethinking Psychometric Evaluation of LLMs: When and Why Self-Reports Predict Behavior","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28345","citing_title":"Auditing LLM-Governed Social Robots with Culture-Specific Moral Gradients","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13339","citing_title":"Probing Persona-Dependent Preferences in Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21564","citing_title":"Measuring Opinion Bias and Sycophancy via LLM-based Persuasion","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04732","citing_title":"Metaphors We Compute By: A Computational Audit of Cultural Translation vs. Thinking in LLMs","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D4YRKDJZLKQVSQUYAXTJA2V3YT","json":"https://pith.science/pith/D4YRKDJZLKQVSQUYAXTJA2V3YT.json","graph_json":"https://pith.science/api/pith-number/D4YRKDJZLKQVSQUYAXTJA2V3YT/graph.json","events_json":"https://pith.science/api/pith-number/D4YRKDJZLKQVSQUYAXTJA2V3YT/events.json","paper":"https://pith.science/paper/D4YRKDJZ"},"agent_actions":{"view_html":"https://pith.science/pith/D4YRKDJZLKQVSQUYAXTJA2V3YT","download_json":"https://pith.science/pith/D4YRKDJZLKQVSQUYAXTJA2V3YT.json","view_paper":"https://pith.science/paper/D4YRKDJZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.08688&json=true","fetch_graph":"https://pith.science/api/pith-number/D4YRKDJZLKQVSQUYAXTJA2V3YT/graph.json","fetch_events":"https://pith.science/api/pith-number/D4YRKDJZLKQVSQUYAXTJA2V3YT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D4YRKDJZLKQVSQUYAXTJA2V3YT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D4YRKDJZLKQVSQUYAXTJA2V3YT/action/storage_attestation","attest_author":"https://pith.science/pith/D4YRKDJZLKQVSQUYAXTJA2V3YT/action/author_attestation","sign_citation":"https://pith.science/pith/D4YRKDJZLKQVSQUYAXTJA2V3YT/action/citation_signature","submit_replication":"https://pith.science/pith/D4YRKDJZLKQVSQUYAXTJA2V3YT/action/replication_record"}},"created_at":"2026-07-05T10:46:23.093846+00:00","updated_at":"2026-07-05T10:46:23.093846+00:00"}