{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3433H2UVIQIR3WF5KH767U73JU","short_pith_number":"pith:3433H2UV","schema_version":"1.0","canonical_sha256":"df37b3ea9544111dd8bd51ffefd3fb4d0c906634ce2b5377b864f666849d140d","source":{"kind":"arxiv","id":"2310.05036","version":3},"attestation_state":"computed","paper":{"title":"AvalonBench: Evaluating LLMs Playing the Game of Avalon","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Jonathan Light, Min Cai, Sheng Shen, Ziniu Hu","submitted_at":"2023-10-08T06:37:08Z","abstract_excerpt":"In this paper, we explore the potential of Large Language Models (LLMs) Agents in playing the strategic social deduction game, Resistance Avalon. Players in Avalon are challenged not only to make informed decisions based on dynamically evolving game phases, but also to engage in discussions where they must deceive, deduce, and negotiate with other players. These characteristics make Avalon a compelling test-bed to study the decision-making and language-processing capabilities of LLM Agents. To facilitate research in this line, we introduce AvalonBench - a comprehensive game environment tailore"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.05036","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-10-08T06:37:08Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"2c09101e49be3d8300a99cba74076b808fb57232d0b2f66fe5ca3442771c5329","abstract_canon_sha256":"b95347cafa3d99135f46ac88b8400eabbd0d88f550d58fd6071e4a3bf5cfdd04"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:10:28.935737Z","signature_b64":"e+UPRPh/JNXr7IPz78d3Y/EfkKcKVpDw0puXlf4buQ4VGV/MU5yPm0oTIj+0Sw3rxJWZ1COwjAQD3zsjn0IMCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"df37b3ea9544111dd8bd51ffefd3fb4d0c906634ce2b5377b864f666849d140d","last_reissued_at":"2026-07-05T07:10:28.935188Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:10:28.935188Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AvalonBench: Evaluating LLMs Playing the Game of Avalon","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Jonathan Light, Min Cai, Sheng Shen, Ziniu Hu","submitted_at":"2023-10-08T06:37:08Z","abstract_excerpt":"In this paper, we explore the potential of Large Language Models (LLMs) Agents in playing the strategic social deduction game, Resistance Avalon. Players in Avalon are challenged not only to make informed decisions based on dynamically evolving game phases, but also to engage in discussions where they must deceive, deduce, and negotiate with other players. These characteristics make Avalon a compelling test-bed to study the decision-making and language-processing capabilities of LLM Agents. To facilitate research in this line, we introduce AvalonBench - a comprehensive game environment tailore"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.05036","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.05036/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.05036","created_at":"2026-07-05T07:10:28.935274+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.05036v3","created_at":"2026-07-05T07:10:28.935274+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.05036","created_at":"2026-07-05T07:10:28.935274+00:00"},{"alias_kind":"pith_short_12","alias_value":"3433H2UVIQIR","created_at":"2026-07-05T07:10:28.935274+00:00"},{"alias_kind":"pith_short_16","alias_value":"3433H2UVIQIR3WF5","created_at":"2026-07-05T07:10:28.935274+00:00"},{"alias_kind":"pith_short_8","alias_value":"3433H2UV","created_at":"2026-07-05T07:10:28.935274+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22748","citing_title":"Agentic World Modeling: Foundations, Capabilities, Laws, and Beyond","ref_index":236,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19308","citing_title":"Enhancing Decision-Making with Large Language Models through Multi-Agent Fictitious Play","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18950","citing_title":"RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19338","citing_title":"Beyond the Current Observation: Evaluating Multimodal Large Language Models in Controllable Non-Markov Games","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12191","citing_title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03544","citing_title":"SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27397","citing_title":"SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27068","citing_title":"QUACK: Questioning, Understanding, and Auditing Communicated Knowledge in Multimodal Social Deduction Agents","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29512","citing_title":"MINDGAMES: A Live Arena for Evaluating Social and Strategic Reasoning in Multi-Agent LLMs","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23238","citing_title":"GENSTRAT: Toward a Science of Strategic Reasoning in Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17510","citing_title":"Scale-Dependent Collective Adaptation in Self-Amending LLM Societies: A Cross-Family Study of Emergent Governance","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2506.17788","citing_title":"Bayesian Social Deduction with Graph-Informed Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13501","citing_title":"A Survey on the Memory Mechanism of Large Language Model based Agents","ref_index":172,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13875","citing_title":"Common-agency Games for Multi-Objective Test-Time Alignment","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2402.01680","citing_title":"Large Language Model based Multi-Agents: A Survey of Progress and Challenges","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25088","citing_title":"Cooperate to Compete: Strategic Coordination in Multi-Agent Conquest","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22748","citing_title":"Agentic World Modeling: Foundations, Capabilities, Laws, and Beyond","ref_index":236,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20582","citing_title":"Trust, Lies, and Long Memories: Emergent Social Dynamics and Reputation in Multi-Round Avalon with LLM Agents","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07003","citing_title":"EmoMAS: Emotion-Aware Multi-Agent System for High-Stakes Edge-Deployable Negotiation with Bayesian Orchestration","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13592","citing_title":"Foresight Optimization for Strategic Reasoning in Large Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16022","citing_title":"SocialGrid: A Benchmark for Planning and Social Reasoning in Embodied Multi-Agent Systems","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20987","citing_title":"Co-Evolving LLM Decision and Skill Bank Agents for Long-Horizon Tasks","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3433H2UVIQIR3WF5KH767U73JU","json":"https://pith.science/pith/3433H2UVIQIR3WF5KH767U73JU.json","graph_json":"https://pith.science/api/pith-number/3433H2UVIQIR3WF5KH767U73JU/graph.json","events_json":"https://pith.science/api/pith-number/3433H2UVIQIR3WF5KH767U73JU/events.json","paper":"https://pith.science/paper/3433H2UV"},"agent_actions":{"view_html":"https://pith.science/pith/3433H2UVIQIR3WF5KH767U73JU","download_json":"https://pith.science/pith/3433H2UVIQIR3WF5KH767U73JU.json","view_paper":"https://pith.science/paper/3433H2UV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.05036&json=true","fetch_graph":"https://pith.science/api/pith-number/3433H2UVIQIR3WF5KH767U73JU/graph.json","fetch_events":"https://pith.science/api/pith-number/3433H2UVIQIR3WF5KH767U73JU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3433H2UVIQIR3WF5KH767U73JU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3433H2UVIQIR3WF5KH767U73JU/action/storage_attestation","attest_author":"https://pith.science/pith/3433H2UVIQIR3WF5KH767U73JU/action/author_attestation","sign_citation":"https://pith.science/pith/3433H2UVIQIR3WF5KH767U73JU/action/citation_signature","submit_replication":"https://pith.science/pith/3433H2UVIQIR3WF5KH767U73JU/action/replication_record"}},"created_at":"2026-07-05T07:10:28.935274+00:00","updated_at":"2026-07-05T07:10:28.935274+00:00"}