{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PDXV53L6SPFOR7VORESKAMTKKZ","short_pith_number":"pith:PDXV53L6","schema_version":"1.0","canonical_sha256":"78ef5eed7e93cae8feae8924a0326a567eb0e3ee8b06c4a66656bdbb2d742af9","source":{"kind":"arxiv","id":"2506.20664","version":1},"attestation_state":"computed","paper":{"title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.HC","cs.MA"],"primary_cat":"cs.AI","authors_text":"Andrei Lupu, Jakob Foerster, Timon Willi","submitted_at":"2025-06-25T17:55:27Z","abstract_excerpt":"As Large Language Models (LLMs) gain agentic abilities, they will have to navigate complex multi-agent scenarios, interacting with human users and other agents in cooperative and competitive settings. This will require new reasoning skills, chief amongst them being theory of mind (ToM), or the ability to reason about the \"mental\" states of other agents. However, ToM and other multi-agent abilities in LLMs are poorly understood, since existing benchmarks suffer from narrow scope, data leakage, saturation, and lack of interactivity. We thus propose Decrypto, a game-based benchmark for multi-agen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.20664","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-25T17:55:27Z","cross_cats_sorted":["cs.CL","cs.HC","cs.MA"],"title_canon_sha256":"ad172432df4728f5a3ead641c7661dde324bb8b03c0469eafc6c6f571ffce751","abstract_canon_sha256":"d3c763691b9514c3043bf5ab2542a11b44456f51f789a795f61deeb6a89089e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:16.122182Z","signature_b64":"GuYjTHfTlvgTc9bnbuASy1OPBJlYIEaptnKuo7cblwB7HMS+zuf75jIEgebLGqAnqvZfyZzD9zCOMU8//7B9Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78ef5eed7e93cae8feae8924a0326a567eb0e3ee8b06c4a66656bdbb2d742af9","last_reissued_at":"2026-07-05T11:27:16.121663Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:16.121663Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.HC","cs.MA"],"primary_cat":"cs.AI","authors_text":"Andrei Lupu, Jakob Foerster, Timon Willi","submitted_at":"2025-06-25T17:55:27Z","abstract_excerpt":"As Large Language Models (LLMs) gain agentic abilities, they will have to navigate complex multi-agent scenarios, interacting with human users and other agents in cooperative and competitive settings. This will require new reasoning skills, chief amongst them being theory of mind (ToM), or the ability to reason about the \"mental\" states of other agents. However, ToM and other multi-agent abilities in LLMs are poorly understood, since existing benchmarks suffer from narrow scope, data leakage, saturation, and lack of interactivity. We thus propose Decrypto, a game-based benchmark for multi-agen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.20664","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.20664/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.20664","created_at":"2026-07-05T11:27:16.121730+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.20664v1","created_at":"2026-07-05T11:27:16.121730+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.20664","created_at":"2026-07-05T11:27:16.121730+00:00"},{"alias_kind":"pith_short_12","alias_value":"PDXV53L6SPFO","created_at":"2026-07-05T11:27:16.121730+00:00"},{"alias_kind":"pith_short_16","alias_value":"PDXV53L6SPFOR7VO","created_at":"2026-07-05T11:27:16.121730+00:00"},{"alias_kind":"pith_short_8","alias_value":"PDXV53L6","created_at":"2026-07-05T11:27:16.121730+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04184","citing_title":"GroupToM-Bench: Benchmarking Group Theory of Mind and Nonlinear Social Emergence in MLLMs","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31916","citing_title":"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PDXV53L6SPFOR7VORESKAMTKKZ","json":"https://pith.science/pith/PDXV53L6SPFOR7VORESKAMTKKZ.json","graph_json":"https://pith.science/api/pith-number/PDXV53L6SPFOR7VORESKAMTKKZ/graph.json","events_json":"https://pith.science/api/pith-number/PDXV53L6SPFOR7VORESKAMTKKZ/events.json","paper":"https://pith.science/paper/PDXV53L6"},"agent_actions":{"view_html":"https://pith.science/pith/PDXV53L6SPFOR7VORESKAMTKKZ","download_json":"https://pith.science/pith/PDXV53L6SPFOR7VORESKAMTKKZ.json","view_paper":"https://pith.science/paper/PDXV53L6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.20664&json=true","fetch_graph":"https://pith.science/api/pith-number/PDXV53L6SPFOR7VORESKAMTKKZ/graph.json","fetch_events":"https://pith.science/api/pith-number/PDXV53L6SPFOR7VORESKAMTKKZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PDXV53L6SPFOR7VORESKAMTKKZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PDXV53L6SPFOR7VORESKAMTKKZ/action/storage_attestation","attest_author":"https://pith.science/pith/PDXV53L6SPFOR7VORESKAMTKKZ/action/author_attestation","sign_citation":"https://pith.science/pith/PDXV53L6SPFOR7VORESKAMTKKZ/action/citation_signature","submit_replication":"https://pith.science/pith/PDXV53L6SPFOR7VORESKAMTKKZ/action/replication_record"}},"created_at":"2026-07-05T11:27:16.121730+00:00","updated_at":"2026-07-05T11:27:16.121730+00:00"}