{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AYCRFTETD36FE5F3X2SISUHKFG","short_pith_number":"pith:AYCRFTET","schema_version":"1.0","canonical_sha256":"060512cc931efc5274bbbea48950ea29a3b03b7704b59796646c33f0883d27a4","source":{"kind":"arxiv","id":"2502.07068","version":2},"attestation_state":"computed","paper":{"title":"Specializing Large Language Models to Simulate Survey Response Distributions for Global Populations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arnav Arora, Daniel Hershcovich, Haijiang Liu, Isabelle Augenstein, Paul R\\\"ottger, Yong Cao","submitted_at":"2025-02-10T21:59:27Z","abstract_excerpt":"Large-scale surveys are essential tools for informing social science research and policy, but running surveys is costly and time-intensive. If we could accurately simulate group-level survey results, this would therefore be very valuable to social science research. Prior work has explored the use of large language models (LLMs) for simulating human behaviors, mostly through prompting. In this paper, we are the first to specialize LLMs for the task of simulating survey response distributions. As a testbed, we use country-level results from two global cultural surveys. We devise a fine-tuning me"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.07068","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-10T21:59:27Z","cross_cats_sorted":[],"title_canon_sha256":"0956f775d4c17878b0b97656c0f249ba94151b60821db6508b8722721d8f62a8","abstract_canon_sha256":"0163c53f73928c1f8465e726430dcf92f4598ed158793f7f1ed66f2580e4bfbd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:00.017272Z","signature_b64":"sqV5hJlw5O+Txiia/f8OYUxkjzDCqFfjdIwCxOfFoWk54mGCViKe/la6XHJSLVzJtrvAGq2/41Ybx7UjxaG4AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"060512cc931efc5274bbbea48950ea29a3b03b7704b59796646c33f0883d27a4","last_reissued_at":"2026-07-05T10:17:00.016792Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:00.016792Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Specializing Large Language Models to Simulate Survey Response Distributions for Global Populations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arnav Arora, Daniel Hershcovich, Haijiang Liu, Isabelle Augenstein, Paul R\\\"ottger, Yong Cao","submitted_at":"2025-02-10T21:59:27Z","abstract_excerpt":"Large-scale surveys are essential tools for informing social science research and policy, but running surveys is costly and time-intensive. If we could accurately simulate group-level survey results, this would therefore be very valuable to social science research. Prior work has explored the use of large language models (LLMs) for simulating human behaviors, mostly through prompting. In this paper, we are the first to specialize LLMs for the task of simulating survey response distributions. As a testbed, we use country-level results from two global cultural surveys. We devise a fine-tuning me"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.07068","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.07068/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.07068","created_at":"2026-07-05T10:17:00.016852+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.07068v2","created_at":"2026-07-05T10:17:00.016852+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.07068","created_at":"2026-07-05T10:17:00.016852+00:00"},{"alias_kind":"pith_short_12","alias_value":"AYCRFTETD36F","created_at":"2026-07-05T10:17:00.016852+00:00"},{"alias_kind":"pith_short_16","alias_value":"AYCRFTETD36FE5F3","created_at":"2026-07-05T10:17:00.016852+00:00"},{"alias_kind":"pith_short_8","alias_value":"AYCRFTET","created_at":"2026-07-05T10:17:00.016852+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12426","citing_title":"Two Wrongs, No Right: Auditing Social-Desirability Bias in LLM Annotators for Computational Social Science","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2502.16761","citing_title":"Language Model Fine-Tuning on Scaled Survey Data for Predicting Distributions of Public Opinions","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09998","citing_title":"Flipping Against All Odds: Reducing LLM Coin Flip Bias via Verbalized Rejection Sampling","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02135","citing_title":"Graph-Based Alternatives to LLMs for Human Simulation","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AYCRFTETD36FE5F3X2SISUHKFG","json":"https://pith.science/pith/AYCRFTETD36FE5F3X2SISUHKFG.json","graph_json":"https://pith.science/api/pith-number/AYCRFTETD36FE5F3X2SISUHKFG/graph.json","events_json":"https://pith.science/api/pith-number/AYCRFTETD36FE5F3X2SISUHKFG/events.json","paper":"https://pith.science/paper/AYCRFTET"},"agent_actions":{"view_html":"https://pith.science/pith/AYCRFTETD36FE5F3X2SISUHKFG","download_json":"https://pith.science/pith/AYCRFTETD36FE5F3X2SISUHKFG.json","view_paper":"https://pith.science/paper/AYCRFTET","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.07068&json=true","fetch_graph":"https://pith.science/api/pith-number/AYCRFTETD36FE5F3X2SISUHKFG/graph.json","fetch_events":"https://pith.science/api/pith-number/AYCRFTETD36FE5F3X2SISUHKFG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AYCRFTETD36FE5F3X2SISUHKFG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AYCRFTETD36FE5F3X2SISUHKFG/action/storage_attestation","attest_author":"https://pith.science/pith/AYCRFTETD36FE5F3X2SISUHKFG/action/author_attestation","sign_citation":"https://pith.science/pith/AYCRFTETD36FE5F3X2SISUHKFG/action/citation_signature","submit_replication":"https://pith.science/pith/AYCRFTETD36FE5F3X2SISUHKFG/action/replication_record"}},"created_at":"2026-07-05T10:17:00.016852+00:00","updated_at":"2026-07-05T10:17:00.016852+00:00"}