{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ENAZONC5UQOJXE22GRFPK3MPFE","short_pith_number":"pith:ENAZONC5","schema_version":"1.0","canonical_sha256":"234197345da41c9b935a344af56d8f290d43758c578f13986529b66b30f94f27","source":{"kind":"arxiv","id":"2209.06899","version":1},"attestation_state":"computed","paper":{"title":"Out of One, Many: Using Language Models to Simulate Human Samples","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Christopher Rytting, David Wingate, Ethan C. Busby, Joshua Gubler, Lisa P. Argyle, Nancy Fulda","submitted_at":"2022-09-14T19:53:32Z","abstract_excerpt":"We propose and explore the possibility that language models can be studied as effective proxies for specific human sub-populations in social science research. Practical and research applications of artificial intelligence tools have sometimes been limited by problematic biases (such as racism or sexism), which are often treated as uniform properties of the models. We show that the \"algorithmic bias\" within one such tool -- the GPT-3 language model -- is instead both fine-grained and demographically correlated, meaning that proper conditioning will cause it to accurately emulate response distri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.06899","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2022-09-14T19:53:32Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"61be5aeadb005e7bfc9969952e217e6716c174f897787eaad490b5853ca6cbe7","abstract_canon_sha256":"687115c7fdf75dbd6a31caa754e4e394ed44f26d7e2afb3ebf9def3de9c71727"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:49:54.541483Z","signature_b64":"2SeyTN1XTWjUpXcKvedyx27Rg64PZuMz37Gt8yRN5H7sMlr/XjtlkSKsXPSwlQIQ0IdTjtRpJyLjfYwx0+ImDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"234197345da41c9b935a344af56d8f290d43758c578f13986529b66b30f94f27","last_reissued_at":"2026-07-05T07:49:54.540503Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:49:54.540503Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Out of One, Many: Using Language Models to Simulate Human Samples","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Christopher Rytting, David Wingate, Ethan C. Busby, Joshua Gubler, Lisa P. Argyle, Nancy Fulda","submitted_at":"2022-09-14T19:53:32Z","abstract_excerpt":"We propose and explore the possibility that language models can be studied as effective proxies for specific human sub-populations in social science research. Practical and research applications of artificial intelligence tools have sometimes been limited by problematic biases (such as racism or sexism), which are often treated as uniform properties of the models. We show that the \"algorithmic bias\" within one such tool -- the GPT-3 language model -- is instead both fine-grained and demographically correlated, meaning that proper conditioning will cause it to accurately emulate response distri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.06899","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.06899/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.06899","created_at":"2026-07-05T07:49:54.541039+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.06899v1","created_at":"2026-07-05T07:49:54.541039+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.06899","created_at":"2026-07-05T07:49:54.541039+00:00"},{"alias_kind":"pith_short_12","alias_value":"ENAZONC5UQOJ","created_at":"2026-07-05T07:49:54.541039+00:00"},{"alias_kind":"pith_short_16","alias_value":"ENAZONC5UQOJXE22","created_at":"2026-07-05T07:49:54.541039+00:00"},{"alias_kind":"pith_short_8","alias_value":"ENAZONC5","created_at":"2026-07-05T07:49:54.541039+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04273","citing_title":"Characterizing initial human-AI proof formalization workflows","ref_index":218,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01045","citing_title":"Child-directed speech facilitates production, not comprehension, in BabyLMs","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12433","citing_title":"Marginal Alignment Does Not Guarantee Joint-Distribution Fidelity: An Official-Reference Audit of Nemotron-Personas-Korea with Cross-Locale Replication","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28802","citing_title":"Human Label Variation as Stable Signal: Learning Annotator-Specific Explanation Behavior via Cross-Annotator Preference Optimization","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06337","citing_title":"Large Language Models as Virtual Survey Respondents: Evaluating Sociodemographic Response Generation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2404.04475","citing_title":"Length-Controlled AlpacaEval: A Simple Way to Debias Automatic Evaluators","ref_index":137,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ENAZONC5UQOJXE22GRFPK3MPFE","json":"https://pith.science/pith/ENAZONC5UQOJXE22GRFPK3MPFE.json","graph_json":"https://pith.science/api/pith-number/ENAZONC5UQOJXE22GRFPK3MPFE/graph.json","events_json":"https://pith.science/api/pith-number/ENAZONC5UQOJXE22GRFPK3MPFE/events.json","paper":"https://pith.science/paper/ENAZONC5"},"agent_actions":{"view_html":"https://pith.science/pith/ENAZONC5UQOJXE22GRFPK3MPFE","download_json":"https://pith.science/pith/ENAZONC5UQOJXE22GRFPK3MPFE.json","view_paper":"https://pith.science/paper/ENAZONC5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.06899&json=true","fetch_graph":"https://pith.science/api/pith-number/ENAZONC5UQOJXE22GRFPK3MPFE/graph.json","fetch_events":"https://pith.science/api/pith-number/ENAZONC5UQOJXE22GRFPK3MPFE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ENAZONC5UQOJXE22GRFPK3MPFE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ENAZONC5UQOJXE22GRFPK3MPFE/action/storage_attestation","attest_author":"https://pith.science/pith/ENAZONC5UQOJXE22GRFPK3MPFE/action/author_attestation","sign_citation":"https://pith.science/pith/ENAZONC5UQOJXE22GRFPK3MPFE/action/citation_signature","submit_replication":"https://pith.science/pith/ENAZONC5UQOJXE22GRFPK3MPFE/action/replication_record"}},"created_at":"2026-07-05T07:49:54.541039+00:00","updated_at":"2026-07-05T07:49:54.541039+00:00"}