{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:EXLHBMIHD4JSBXULQIRO3YIUHJ","short_pith_number":"pith:EXLHBMIH","schema_version":"1.0","canonical_sha256":"25d670b1071f1320de8b8222ede1143a5c04282c55efd151ccd8d6d307621e78","source":{"kind":"arxiv","id":"2606.19336","version":1},"attestation_state":"computed","paper":{"title":"Learning User Simulators with Turing Rewards","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alex Pentland, Cedegao E. Zhang, Linlu Qiu, Pengyuan Li, Roger P. Levy, Yingshan Susan Wang, Yoon Kim, Zexue He","submitted_at":"2026-06-17T17:58:48Z","abstract_excerpt":"Learning to simulate human users in interactive settings could advance the training of agent assistants, evaluation of personalization systems, research in the social sciences, and more. Existing approaches generally do so by training a large language model (LLM) to match a single ground truth response, either by maximizing the log probability or by using a similarity reward. We instead propose {Turing-RL}: a Turing-Test-based reinforcement learning approach for training user simulator models. {Turing-RL} uses a discriminative Turing reward with an LLM judge to score how indistinguishable a ge"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2606.19336","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2026-06-17T17:58:48Z","cross_cats_sorted":[],"title_canon_sha256":"02e6ea577816a1b91a5f2a343005a571f11f067779607decf4765f6d72bfb965","abstract_canon_sha256":"869e377d20f0f50cac038078619f8c2f70b587e8cbd426334b3ba7a9e9ba6194"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-19T16:12:11.518963Z","signature_b64":"8/7eOJgSngBZxMJa8B2Px8PRBcujiymNVkFiFgtZBQKOkwUWDeJ003zx8HXcY5kPL5+eIwdtCrQYmIWtgaOqCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"25d670b1071f1320de8b8222ede1143a5c04282c55efd151ccd8d6d307621e78","last_reissued_at":"2026-06-19T16:12:11.518627Z","signature_status":"signed_v1","first_computed_at":"2026-06-19T16:12:11.518627Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning User Simulators with Turing Rewards","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alex Pentland, Cedegao E. Zhang, Linlu Qiu, Pengyuan Li, Roger P. Levy, Yingshan Susan Wang, Yoon Kim, Zexue He","submitted_at":"2026-06-17T17:58:48Z","abstract_excerpt":"Learning to simulate human users in interactive settings could advance the training of agent assistants, evaluation of personalization systems, research in the social sciences, and more. Existing approaches generally do so by training a large language model (LLM) to match a single ground truth response, either by maximizing the log probability or by using a similarity reward. We instead propose {Turing-RL}: a Turing-Test-based reinforcement learning approach for training user simulator models. {Turing-RL} uses a discriminative Turing reward with an LLM judge to score how indistinguishable a ge"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.19336","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2606.19336/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2606.19336","created_at":"2026-06-19T16:12:11.518687+00:00"},{"alias_kind":"arxiv_version","alias_value":"2606.19336v1","created_at":"2026-06-19T16:12:11.518687+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.19336","created_at":"2026-06-19T16:12:11.518687+00:00"},{"alias_kind":"pith_short_12","alias_value":"EXLHBMIHD4JS","created_at":"2026-06-19T16:12:11.518687+00:00"},{"alias_kind":"pith_short_16","alias_value":"EXLHBMIHD4JSBXUL","created_at":"2026-06-19T16:12:11.518687+00:00"},{"alias_kind":"pith_short_8","alias_value":"EXLHBMIH","created_at":"2026-06-19T16:12:11.518687+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.09420","citing_title":"Intent Speaks Louder: Controllable User Simulation Beyond Response Imitation","ref_index":2025,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EXLHBMIHD4JSBXULQIRO3YIUHJ","json":"https://pith.science/pith/EXLHBMIHD4JSBXULQIRO3YIUHJ.json","graph_json":"https://pith.science/api/pith-number/EXLHBMIHD4JSBXULQIRO3YIUHJ/graph.json","events_json":"https://pith.science/api/pith-number/EXLHBMIHD4JSBXULQIRO3YIUHJ/events.json","paper":"https://pith.science/paper/EXLHBMIH"},"agent_actions":{"view_html":"https://pith.science/pith/EXLHBMIHD4JSBXULQIRO3YIUHJ","download_json":"https://pith.science/pith/EXLHBMIHD4JSBXULQIRO3YIUHJ.json","view_paper":"https://pith.science/paper/EXLHBMIH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2606.19336&json=true","fetch_graph":"https://pith.science/api/pith-number/EXLHBMIHD4JSBXULQIRO3YIUHJ/graph.json","fetch_events":"https://pith.science/api/pith-number/EXLHBMIHD4JSBXULQIRO3YIUHJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EXLHBMIHD4JSBXULQIRO3YIUHJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EXLHBMIHD4JSBXULQIRO3YIUHJ/action/storage_attestation","attest_author":"https://pith.science/pith/EXLHBMIHD4JSBXULQIRO3YIUHJ/action/author_attestation","sign_citation":"https://pith.science/pith/EXLHBMIHD4JSBXULQIRO3YIUHJ/action/citation_signature","submit_replication":"https://pith.science/pith/EXLHBMIHD4JSBXULQIRO3YIUHJ/action/replication_record"}},"created_at":"2026-06-19T16:12:11.518687+00:00","updated_at":"2026-06-19T16:12:11.518687+00:00"}