{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:35TESAL5TZYZIKB2746EZ72Q62","short_pith_number":"pith:35TESAL5","schema_version":"1.0","canonical_sha256":"df6649017d9e7194283aff3c4cff50f6a817a297a01344e01f002b7c53a8b136","source":{"kind":"arxiv","id":"2310.18940","version":4},"attestation_state":"computed","paper":{"title":"Language Agents with Reinforcement Learning for Strategic Play in the Werewolf Game","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"cs.AI","authors_text":"Chao Yu, Fei Fang, Yi Wu, Yu Wang, Zelai Xu","submitted_at":"2023-10-29T09:02:57Z","abstract_excerpt":"Agents built with large language models (LLMs) have shown great potential across a wide range of domains. However, in complex decision-making tasks, pure LLM-based agents tend to exhibit intrinsic bias in their choice of actions, which is inherited from the model's training data and results in suboptimal performance. To develop strategic language agents, i.e., agents that generate flexible language actions and possess strong decision-making abilities, we propose a novel framework that powers LLM-based agents with reinforcement learning (RL). We consider Werewolf, a popular social deduction gam"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.18940","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-10-29T09:02:57Z","cross_cats_sorted":["cs.LG","cs.MA"],"title_canon_sha256":"4f484f353352a7426b02b4caf8ac92c67bcf7ba7d88cfac8c1f231ce9d5c29d6","abstract_canon_sha256":"69ed2284114fc30edc7b2c559b60029ec3b95cedaf2a6a3f2cc622a5fe2b2637"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:30.425473Z","signature_b64":"/b3qiBCQvBkJFDltxxDk+4vsUv6CL5hW5idOy/ju/v/6S7UZSPVREk6PnbquncdNayNjEdVDKGfQHbk9ujiCBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"df6649017d9e7194283aff3c4cff50f6a817a297a01344e01f002b7c53a8b136","last_reissued_at":"2026-07-05T11:11:30.424929Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:30.424929Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language Agents with Reinforcement Learning for Strategic Play in the Werewolf Game","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"cs.AI","authors_text":"Chao Yu, Fei Fang, Yi Wu, Yu Wang, Zelai Xu","submitted_at":"2023-10-29T09:02:57Z","abstract_excerpt":"Agents built with large language models (LLMs) have shown great potential across a wide range of domains. However, in complex decision-making tasks, pure LLM-based agents tend to exhibit intrinsic bias in their choice of actions, which is inherited from the model's training data and results in suboptimal performance. To develop strategic language agents, i.e., agents that generate flexible language actions and possess strong decision-making abilities, we propose a novel framework that powers LLM-based agents with reinforcement learning (RL). We consider Werewolf, a popular social deduction gam"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.18940","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.18940/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.18940","created_at":"2026-07-05T11:11:30.424988+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.18940v4","created_at":"2026-07-05T11:11:30.424988+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.18940","created_at":"2026-07-05T11:11:30.424988+00:00"},{"alias_kind":"pith_short_12","alias_value":"35TESAL5TZYZ","created_at":"2026-07-05T11:11:30.424988+00:00"},{"alias_kind":"pith_short_16","alias_value":"35TESAL5TZYZIKB2","created_at":"2026-07-05T11:11:30.424988+00:00"},{"alias_kind":"pith_short_8","alias_value":"35TESAL5","created_at":"2026-07-05T11:11:30.424988+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13608","citing_title":"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02754","citing_title":"$\\Psi$-Bench: Evaluating Persona-Sensitive Influencing in Persuasive Dialogues","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27397","citing_title":"SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27068","citing_title":"QUACK: Questioning, Understanding, and Auditing Communicated Knowledge in Multimodal Social Deduction Agents","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29512","citing_title":"MINDGAMES: A Live Arena for Evaluating Social and Strategic Reasoning in Multi-Agent LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":237,"is_internal_anchor":false},{"citing_arxiv_id":"2506.02546","citing_title":"To trust or not to trust: Attention-based Trust Management for LLM Multi-Agent Systems","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23023","citing_title":"Deceive, Detect, and Disclose: Large Language Models Play Mini-Mafia","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2402.01680","citing_title":"Large Language Model based Multi-Agents: A Survey of Progress and Challenges","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07301","citing_title":"SOM: Structured Opponent Modeling for LLM-based Agents via Structural Causal Model","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/35TESAL5TZYZIKB2746EZ72Q62","json":"https://pith.science/pith/35TESAL5TZYZIKB2746EZ72Q62.json","graph_json":"https://pith.science/api/pith-number/35TESAL5TZYZIKB2746EZ72Q62/graph.json","events_json":"https://pith.science/api/pith-number/35TESAL5TZYZIKB2746EZ72Q62/events.json","paper":"https://pith.science/paper/35TESAL5"},"agent_actions":{"view_html":"https://pith.science/pith/35TESAL5TZYZIKB2746EZ72Q62","download_json":"https://pith.science/pith/35TESAL5TZYZIKB2746EZ72Q62.json","view_paper":"https://pith.science/paper/35TESAL5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.18940&json=true","fetch_graph":"https://pith.science/api/pith-number/35TESAL5TZYZIKB2746EZ72Q62/graph.json","fetch_events":"https://pith.science/api/pith-number/35TESAL5TZYZIKB2746EZ72Q62/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/35TESAL5TZYZIKB2746EZ72Q62/action/timestamp_anchor","attest_storage":"https://pith.science/pith/35TESAL5TZYZIKB2746EZ72Q62/action/storage_attestation","attest_author":"https://pith.science/pith/35TESAL5TZYZIKB2746EZ72Q62/action/author_attestation","sign_citation":"https://pith.science/pith/35TESAL5TZYZIKB2746EZ72Q62/action/citation_signature","submit_replication":"https://pith.science/pith/35TESAL5TZYZIKB2746EZ72Q62/action/replication_record"}},"created_at":"2026-07-05T11:11:30.424988+00:00","updated_at":"2026-07-05T11:11:30.424988+00:00"}