{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:TFZXCHBFNJ3LBZGRZPDHFNDY37","short_pith_number":"pith:TFZXCHBF","schema_version":"1.0","canonical_sha256":"9973711c256a76b0e4d1cbc672b478dfec7b685d4d2d110cdd0727634fd43c2b","source":{"kind":"arxiv","id":"2603.11245","version":2},"attestation_state":"computed","paper":{"title":"Mind the Sim2Real Gap in User Simulation for Agentic Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Graham Neubig, Jiarui Liu, Maarten Sap, Qianou Ma, Sean Welleck, Sherry Tongshuang Wu, Weihua Du, Weiwei Sun, Xuhui Zhou, Yiming Yang, Yiqing Xie","submitted_at":"2026-03-11T19:12:31Z","abstract_excerpt":"As NLP evaluation shifts from static benchmarks to multi-turn interactive settings, LLM-based simulators have become widely used as user proxies, serving two roles: generating user turns and providing evaluation signals. Yet, these simulations are frequently assumed to be faithful to real human behaviors, often without rigorous verification. We formalize the Sim2Real gap in user simulation and present the first study running the full $\\tau$-bench protocol with real humans (451 participants, 165 tasks), benchmarking 31 LLM simulators across proprietary, open-source, and specialized families usi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2603.11245","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-03-11T19:12:31Z","cross_cats_sorted":[],"title_canon_sha256":"8f3fa3eb5371aa2d61f095d3e48a630e9bd2f4e40dfe735b7769c72060ab5a35","abstract_canon_sha256":"4f54ecb8d54ee858e74af66ff7f4678088798510f5bf83fdf9213eee4efc427d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T00:31:13.338042Z","signature_b64":"ad5BVgGkdogCwGnJMJ0ocz1Cj/HYFiAGwAFXThiZYt24OPB1aZuLrnQKYpEMw7LnotXGQNr0Vg3zLeAChRZ0CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9973711c256a76b0e4d1cbc672b478dfec7b685d4d2d110cdd0727634fd43c2b","last_reissued_at":"2026-08-04T00:31:13.336239Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T00:31:13.336239Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mind the Sim2Real Gap in User Simulation for Agentic Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Graham Neubig, Jiarui Liu, Maarten Sap, Qianou Ma, Sean Welleck, Sherry Tongshuang Wu, Weihua Du, Weiwei Sun, Xuhui Zhou, Yiming Yang, Yiqing Xie","submitted_at":"2026-03-11T19:12:31Z","abstract_excerpt":"As NLP evaluation shifts from static benchmarks to multi-turn interactive settings, LLM-based simulators have become widely used as user proxies, serving two roles: generating user turns and providing evaluation signals. Yet, these simulations are frequently assumed to be faithful to real human behaviors, often without rigorous verification. We formalize the Sim2Real gap in user simulation and present the first study running the full $\\tau$-bench protocol with real humans (451 participants, 165 tasks), benchmarking 31 LLM simulators across proprietary, open-source, and specialized families usi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2603.11245","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2603.11245/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2603.11245","created_at":"2026-08-04T00:31:13.337365+00:00"},{"alias_kind":"arxiv_version","alias_value":"2603.11245v2","created_at":"2026-08-04T00:31:13.337365+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2603.11245","created_at":"2026-08-04T00:31:13.337365+00:00"},{"alias_kind":"pith_short_12","alias_value":"TFZXCHBFNJ3L","created_at":"2026-08-04T00:31:13.337365+00:00"},{"alias_kind":"pith_short_16","alias_value":"TFZXCHBFNJ3LBZGR","created_at":"2026-08-04T00:31:13.337365+00:00"},{"alias_kind":"pith_short_8","alias_value":"TFZXCHBF","created_at":"2026-08-04T00:31:13.337365+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":13,"sample":[{"citing_arxiv_id":"2607.06993","citing_title":"Large Behavior Model: A Promptable Digital Twin of the Retail Customer","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20708","citing_title":"Simulated Customers Never Walk Away: Decision Fidelity of LLM User Simulators Measured Against Real Purchase Outcomes","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12969","citing_title":"Multi-Modal Agents for Power Distribution Defect Detection: An Evaluation of Foundation Models","ref_index":44,"is_internal_anchor":true},{"citing_arxiv_id":"2607.02464","citing_title":"Will Scaling Improve Social Simulation with LLMs?","ref_index":45,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11520","citing_title":"ISE: An Execution-Grounded Recipe for Multi-Turn OS-Agent Trajectories","ref_index":69,"is_internal_anchor":true},{"citing_arxiv_id":"2606.02798","citing_title":"BehaviorBench: Modeling Real-World User Decisions from Behavioral Traces","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2605.13841","citing_title":"EVA-Bench: A New End-to-end Framework for Evaluating Voice Agents","ref_index":39,"is_internal_anchor":true},{"citing_arxiv_id":"2605.25680","citing_title":"Simulating Human Memory with Language Models","ref_index":59,"is_internal_anchor":true},{"citing_arxiv_id":"2605.20506","citing_title":"Reinforcing Human Behavior Simulation via Verbal Feedback","ref_index":63,"is_internal_anchor":true},{"citing_arxiv_id":"2605.13841","citing_title":"EVA-Bench: A New End-to-end Framework for Evaluating Voice Agents","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2604.02315","citing_title":"Beyond the Assistant Turn: User Turn Generation as a Probe of Interaction Awareness in Language Models","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2605.09808","citing_title":"Quantifying the Utility of User Simulators for Building Collaborative LLM Assistants","ref_index":29,"is_internal_anchor":true},{"citing_arxiv_id":"2605.05700","citing_title":"An Empirical Study of Proactive Coding Assistants in Real-World Software Development","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TFZXCHBFNJ3LBZGRZPDHFNDY37","json":"https://pith.science/pith/TFZXCHBFNJ3LBZGRZPDHFNDY37.json","graph_json":"https://pith.science/api/pith-number/TFZXCHBFNJ3LBZGRZPDHFNDY37/graph.json","events_json":"https://pith.science/api/pith-number/TFZXCHBFNJ3LBZGRZPDHFNDY37/events.json","paper":"https://pith.science/paper/TFZXCHBF"},"agent_actions":{"view_html":"https://pith.science/pith/TFZXCHBFNJ3LBZGRZPDHFNDY37","download_json":"https://pith.science/pith/TFZXCHBFNJ3LBZGRZPDHFNDY37.json","view_paper":"https://pith.science/paper/TFZXCHBF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2603.11245&json=true","fetch_graph":"https://pith.science/api/pith-number/TFZXCHBFNJ3LBZGRZPDHFNDY37/graph.json","fetch_events":"https://pith.science/api/pith-number/TFZXCHBFNJ3LBZGRZPDHFNDY37/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TFZXCHBFNJ3LBZGRZPDHFNDY37/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TFZXCHBFNJ3LBZGRZPDHFNDY37/action/storage_attestation","attest_author":"https://pith.science/pith/TFZXCHBFNJ3LBZGRZPDHFNDY37/action/author_attestation","sign_citation":"https://pith.science/pith/TFZXCHBFNJ3LBZGRZPDHFNDY37/action/citation_signature","submit_replication":"https://pith.science/pith/TFZXCHBFNJ3LBZGRZPDHFNDY37/action/replication_record"}},"created_at":"2026-08-04T00:31:13.337365+00:00","updated_at":"2026-08-04T00:31:13.337365+00:00"}