{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:V7KVW5NITVNQBH3CUJRNDBD4ZI","short_pith_number":"pith:V7KVW5NI","schema_version":"1.0","canonical_sha256":"afd55b75a89d5b009f62a262d1847cca19e8722614b8bdece8a0f08ef5a91096","source":{"kind":"arxiv","id":"2507.16196","version":1},"attestation_state":"computed","paper":{"title":"Do Large Language Models Have a Planning Theory of Mind? Evidence from MindGames: a Multi-Step Persuasion Task","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Beba Cibralic, Cameron R. Jones, Jared Moore, Ned Cooper, Nick Haber, Rasmus Overmark","submitted_at":"2025-07-22T03:15:27Z","abstract_excerpt":"Recent evidence suggests Large Language Models (LLMs) display Theory of Mind (ToM) abilities. Most ToM experiments place participants in a spectatorial role, wherein they predict and interpret other agents' behavior. However, human ToM also contributes to dynamically planning action and strategically intervening on others' mental states. We present MindGames: a novel `planning theory of mind' (PToM) task which requires agents to infer an interlocutor's beliefs and desires to persuade them to alter their behavior. Unlike previous evaluations, we explicitly evaluate use cases of ToM. We find tha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.16196","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-22T03:15:27Z","cross_cats_sorted":[],"title_canon_sha256":"0e2b90ed0b30c5239de69ed552bd517626857f081aa3ab6d749874818ad46396","abstract_canon_sha256":"fb81e90bb11c9ee70557dc650431e803f22706fa3f47537313490ff74985b460"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:41:05.044115Z","signature_b64":"C1Ob1tGiVg8O0ICkj6OHcnbSOPKHW2IZ6C6zG+44ey1nfj23wkQUL7OER/g/QE1XOfZK3QiTRXYqoul3FnIeDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"afd55b75a89d5b009f62a262d1847cca19e8722614b8bdece8a0f08ef5a91096","last_reissued_at":"2026-07-05T11:41:05.043684Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:41:05.043684Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Large Language Models Have a Planning Theory of Mind? Evidence from MindGames: a Multi-Step Persuasion Task","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Beba Cibralic, Cameron R. Jones, Jared Moore, Ned Cooper, Nick Haber, Rasmus Overmark","submitted_at":"2025-07-22T03:15:27Z","abstract_excerpt":"Recent evidence suggests Large Language Models (LLMs) display Theory of Mind (ToM) abilities. Most ToM experiments place participants in a spectatorial role, wherein they predict and interpret other agents' behavior. However, human ToM also contributes to dynamically planning action and strategically intervening on others' mental states. We present MindGames: a novel `planning theory of mind' (PToM) task which requires agents to infer an interlocutor's beliefs and desires to persuade them to alter their behavior. Unlike previous evaluations, we explicitly evaluate use cases of ToM. We find tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.16196","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.16196/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.16196","created_at":"2026-07-05T11:41:05.043741+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.16196v1","created_at":"2026-07-05T11:41:05.043741+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.16196","created_at":"2026-07-05T11:41:05.043741+00:00"},{"alias_kind":"pith_short_12","alias_value":"V7KVW5NITVNQ","created_at":"2026-07-05T11:41:05.043741+00:00"},{"alias_kind":"pith_short_16","alias_value":"V7KVW5NITVNQBH3C","created_at":"2026-07-05T11:41:05.043741+00:00"},{"alias_kind":"pith_short_8","alias_value":"V7KVW5NI","created_at":"2026-07-05T11:41:05.043741+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31916","citing_title":"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05330","citing_title":"A Model of Multi-turn Human Persuadability Using Probabilistic Belief Tracing","ref_index":90,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V7KVW5NITVNQBH3CUJRNDBD4ZI","json":"https://pith.science/pith/V7KVW5NITVNQBH3CUJRNDBD4ZI.json","graph_json":"https://pith.science/api/pith-number/V7KVW5NITVNQBH3CUJRNDBD4ZI/graph.json","events_json":"https://pith.science/api/pith-number/V7KVW5NITVNQBH3CUJRNDBD4ZI/events.json","paper":"https://pith.science/paper/V7KVW5NI"},"agent_actions":{"view_html":"https://pith.science/pith/V7KVW5NITVNQBH3CUJRNDBD4ZI","download_json":"https://pith.science/pith/V7KVW5NITVNQBH3CUJRNDBD4ZI.json","view_paper":"https://pith.science/paper/V7KVW5NI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.16196&json=true","fetch_graph":"https://pith.science/api/pith-number/V7KVW5NITVNQBH3CUJRNDBD4ZI/graph.json","fetch_events":"https://pith.science/api/pith-number/V7KVW5NITVNQBH3CUJRNDBD4ZI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V7KVW5NITVNQBH3CUJRNDBD4ZI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V7KVW5NITVNQBH3CUJRNDBD4ZI/action/storage_attestation","attest_author":"https://pith.science/pith/V7KVW5NITVNQBH3CUJRNDBD4ZI/action/author_attestation","sign_citation":"https://pith.science/pith/V7KVW5NITVNQBH3CUJRNDBD4ZI/action/citation_signature","submit_replication":"https://pith.science/pith/V7KVW5NITVNQBH3CUJRNDBD4ZI/action/replication_record"}},"created_at":"2026-07-05T11:41:05.043741+00:00","updated_at":"2026-07-05T11:41:05.043741+00:00"}