{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ENSINSLS37WDA4KENR64I4JX3O","short_pith_number":"pith:ENSINSLS","schema_version":"1.0","canonical_sha256":"236486c972dfec3071446c7dc47137dbbf26987a254ba5c80aaeaca853c59851","source":{"kind":"arxiv","id":"2306.07951","version":4},"attestation_state":"computed","paper":{"title":"Questioning the Survey Responses of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Celestine Mendler-D\\\"unner, Moritz Hardt, Ricardo Dominguez-Olmedo","submitted_at":"2023-06-13T17:48:27Z","abstract_excerpt":"Surveys have recently gained popularity as a tool to study large language models. By comparing survey responses of models to those of human reference populations, researchers aim to infer the demographics, political opinions, or values best represented by current language models. In this work, we critically examine this methodology on the basis of the well-established American Community Survey by the U.S. Census Bureau. Evaluating 43 different language models using de-facto standard prompting methodologies, we establish two dominant patterns. First, models' responses are governed by ordering a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.07951","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-13T17:48:27Z","cross_cats_sorted":[],"title_canon_sha256":"7a86b194c7df74acb350e2769ee0ef61c09ae54ebdcd9f844576fd064e1d07df","abstract_canon_sha256":"d399589be8b53c129285dce6b2fcca6ef180d793bfc2b476998b64ba4a8c9eb2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:40.332468Z","signature_b64":"gidk6cveoeDzNnmqbrP/IL/VK3aWNomwtNPr+sLvpFf7rFk0Om7PHs8x9+jWYJ/51Xs8ls+jSRph1kqXLu4sBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"236486c972dfec3071446c7dc47137dbbf26987a254ba5c80aaeaca853c59851","last_reissued_at":"2026-07-05T09:45:40.331926Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:40.331926Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Questioning the Survey Responses of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Celestine Mendler-D\\\"unner, Moritz Hardt, Ricardo Dominguez-Olmedo","submitted_at":"2023-06-13T17:48:27Z","abstract_excerpt":"Surveys have recently gained popularity as a tool to study large language models. By comparing survey responses of models to those of human reference populations, researchers aim to infer the demographics, political opinions, or values best represented by current language models. In this work, we critically examine this methodology on the basis of the well-established American Community Survey by the U.S. Census Bureau. Evaluating 43 different language models using de-facto standard prompting methodologies, we establish two dominant patterns. First, models' responses are governed by ordering a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.07951","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.07951/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.07951","created_at":"2026-07-05T09:45:40.331992+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.07951v4","created_at":"2026-07-05T09:45:40.331992+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.07951","created_at":"2026-07-05T09:45:40.331992+00:00"},{"alias_kind":"pith_short_12","alias_value":"ENSINSLS37WD","created_at":"2026-07-05T09:45:40.331992+00:00"},{"alias_kind":"pith_short_16","alias_value":"ENSINSLS37WDA4KE","created_at":"2026-07-05T09:45:40.331992+00:00"},{"alias_kind":"pith_short_8","alias_value":"ENSINSLS","created_at":"2026-07-05T09:45:40.331992+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12433","citing_title":"Marginal Alignment Does Not Guarantee Joint-Distribution Fidelity: An Official-Reference Audit of Nemotron-Personas-Korea with Cross-Locale Replication","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30085","citing_title":"Not-quite-human tastes: the stylized omnivorousness of LLM survey surrogates","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2305.09620","citing_title":"AI-Augmented Surveys: Leveraging Large Language Models and Surveys for Opinion Prediction","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2507.23009","citing_title":"Position: Stop Evaluating AI with Human Tests, Develop Principled, AI-specific Tests instead","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02135","citing_title":"Graph-Based Alternatives to LLMs for Human Simulation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01846","citing_title":"Do Large Language Models Plan Answer Positions? Position Bias in Multiple-Choice Question Generation","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ENSINSLS37WDA4KENR64I4JX3O","json":"https://pith.science/pith/ENSINSLS37WDA4KENR64I4JX3O.json","graph_json":"https://pith.science/api/pith-number/ENSINSLS37WDA4KENR64I4JX3O/graph.json","events_json":"https://pith.science/api/pith-number/ENSINSLS37WDA4KENR64I4JX3O/events.json","paper":"https://pith.science/paper/ENSINSLS"},"agent_actions":{"view_html":"https://pith.science/pith/ENSINSLS37WDA4KENR64I4JX3O","download_json":"https://pith.science/pith/ENSINSLS37WDA4KENR64I4JX3O.json","view_paper":"https://pith.science/paper/ENSINSLS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.07951&json=true","fetch_graph":"https://pith.science/api/pith-number/ENSINSLS37WDA4KENR64I4JX3O/graph.json","fetch_events":"https://pith.science/api/pith-number/ENSINSLS37WDA4KENR64I4JX3O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ENSINSLS37WDA4KENR64I4JX3O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ENSINSLS37WDA4KENR64I4JX3O/action/storage_attestation","attest_author":"https://pith.science/pith/ENSINSLS37WDA4KENR64I4JX3O/action/author_attestation","sign_citation":"https://pith.science/pith/ENSINSLS37WDA4KENR64I4JX3O/action/citation_signature","submit_replication":"https://pith.science/pith/ENSINSLS37WDA4KENR64I4JX3O/action/replication_record"}},"created_at":"2026-07-05T09:45:40.331992+00:00","updated_at":"2026-07-05T09:45:40.331992+00:00"}