{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SRWEO5F5QI36BXDABS7O27Y33J","short_pith_number":"pith:SRWEO5F5","schema_version":"1.0","canonical_sha256":"946c4774bd8237e0dc600cbeed7f1bda69bf8c6d26486bd65ebb1bea4845333e","source":{"kind":"arxiv","id":"2502.06060","version":1},"attestation_state":"computed","paper":{"title":"Training Language Models for Social Deduction with Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.MA"],"primary_cat":"cs.AI","authors_text":"Bidipta Sarkar, C. Karen Liu, Dorsa Sadigh, Warren Xia","submitted_at":"2025-02-09T22:44:45Z","abstract_excerpt":"Communicating in natural language is a powerful tool in multi-agent settings, as it enables independent agents to share information in partially observable settings and allows zero-shot coordination with humans. However, most prior works are limited as they either rely on training with large amounts of human demonstrations or lack the ability to generate natural and useful communication strategies. In this work, we train language models to have productive discussions about their environment in natural language without any human demonstrations. We decompose the communication problem into listen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06060","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-02-09T22:44:45Z","cross_cats_sorted":["cs.CL","cs.LG","cs.MA"],"title_canon_sha256":"74a36db7748ad47cfca06e12268d4d47d5f028bcbeb473192bf56437edce4e1b","abstract_canon_sha256":"a27fbfbf3d031abd0c4ea2f5c205138568917df1d65c66e19cf19e5bda4e6c5a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:47.112376Z","signature_b64":"M5gfubHCWlUM++AgRxW6Cqz+mF7T6h3l9kdnr/hWduS3wbl/+pB0+ZU5u4nlMaJWPdk5cYx8D9AjE5c00UlXDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"946c4774bd8237e0dc600cbeed7f1bda69bf8c6d26486bd65ebb1bea4845333e","last_reissued_at":"2026-07-05T10:11:47.111889Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:47.111889Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Language Models for Social Deduction with Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.MA"],"primary_cat":"cs.AI","authors_text":"Bidipta Sarkar, C. Karen Liu, Dorsa Sadigh, Warren Xia","submitted_at":"2025-02-09T22:44:45Z","abstract_excerpt":"Communicating in natural language is a powerful tool in multi-agent settings, as it enables independent agents to share information in partially observable settings and allows zero-shot coordination with humans. However, most prior works are limited as they either rely on training with large amounts of human demonstrations or lack the ability to generate natural and useful communication strategies. In this work, we train language models to have productive discussions about their environment in natural language without any human demonstrations. We decompose the communication problem into listen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06060","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06060/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06060","created_at":"2026-07-05T10:11:47.111942+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06060v1","created_at":"2026-07-05T10:11:47.111942+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06060","created_at":"2026-07-05T10:11:47.111942+00:00"},{"alias_kind":"pith_short_12","alias_value":"SRWEO5F5QI36","created_at":"2026-07-05T10:11:47.111942+00:00"},{"alias_kind":"pith_short_16","alias_value":"SRWEO5F5QI36BXDA","created_at":"2026-07-05T10:11:47.111942+00:00"},{"alias_kind":"pith_short_8","alias_value":"SRWEO5F5","created_at":"2026-07-05T10:11:47.111942+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.17788","citing_title":"Bayesian Social Deduction with Graph-Informed Language Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21972","citing_title":"Learning Decentralized LLM Collaboration with Multi-Agent Actor Critic","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SRWEO5F5QI36BXDABS7O27Y33J","json":"https://pith.science/pith/SRWEO5F5QI36BXDABS7O27Y33J.json","graph_json":"https://pith.science/api/pith-number/SRWEO5F5QI36BXDABS7O27Y33J/graph.json","events_json":"https://pith.science/api/pith-number/SRWEO5F5QI36BXDABS7O27Y33J/events.json","paper":"https://pith.science/paper/SRWEO5F5"},"agent_actions":{"view_html":"https://pith.science/pith/SRWEO5F5QI36BXDABS7O27Y33J","download_json":"https://pith.science/pith/SRWEO5F5QI36BXDABS7O27Y33J.json","view_paper":"https://pith.science/paper/SRWEO5F5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06060&json=true","fetch_graph":"https://pith.science/api/pith-number/SRWEO5F5QI36BXDABS7O27Y33J/graph.json","fetch_events":"https://pith.science/api/pith-number/SRWEO5F5QI36BXDABS7O27Y33J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SRWEO5F5QI36BXDABS7O27Y33J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SRWEO5F5QI36BXDABS7O27Y33J/action/storage_attestation","attest_author":"https://pith.science/pith/SRWEO5F5QI36BXDABS7O27Y33J/action/author_attestation","sign_citation":"https://pith.science/pith/SRWEO5F5QI36BXDABS7O27Y33J/action/citation_signature","submit_replication":"https://pith.science/pith/SRWEO5F5QI36BXDABS7O27Y33J/action/replication_record"}},"created_at":"2026-07-05T10:11:47.111942+00:00","updated_at":"2026-07-05T10:11:47.111942+00:00"}