{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WSG62FUGLCKNM2NF4SHLYNJPO3","short_pith_number":"pith:WSG62FUG","schema_version":"1.0","canonical_sha256":"b48ded16865894d669a5e48ebc352f76c7de2b88cc0a7c12d1638887ae6b7090","source":{"kind":"arxiv","id":"2412.12175","version":1},"attestation_state":"computed","paper":{"title":"Explore Theory of Mind: Program-guided adversarial data generation for theory of mind reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Asli Celikyilmaz, Jane Yu, Maryam Fazel-Zarandi, Melanie Sclar, Yejin Choi, Yonatan Bisk, Yulia Tsvetkov","submitted_at":"2024-12-12T21:29:00Z","abstract_excerpt":"Do large language models (LLMs) have theory of mind? A plethora of papers and benchmarks have been introduced to evaluate if current models have been able to develop this key ability of social intelligence. However, all rely on limited datasets with simple patterns that can potentially lead to problematic blind spots in evaluation and an overestimation of model capabilities. We introduce ExploreToM, the first framework to allow large-scale generation of diverse and challenging theory of mind data for robust training and evaluation. Our approach leverages an A* search over a custom domain-speci"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.12175","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-12T21:29:00Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"94a19954a7f6e9569b2eebb0b426d55c4bd2912f26efb314b30e018651aef2de","abstract_canon_sha256":"e10bd844581a9c48d97859ca5237afa24a00baadb0284945a3d81b4871cba339"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:50:11.079791Z","signature_b64":"yte40uQfPvmqDkpc+3FCQ+3sBgg7AYZ3MiroNVlw6i9KQSyIiJtgmYNySSZUoSKwTuQIe8TXNOPesT2UseHuAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b48ded16865894d669a5e48ebc352f76c7de2b88cc0a7c12d1638887ae6b7090","last_reissued_at":"2026-07-05T09:50:11.079323Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:50:11.079323Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Explore Theory of Mind: Program-guided adversarial data generation for theory of mind reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Asli Celikyilmaz, Jane Yu, Maryam Fazel-Zarandi, Melanie Sclar, Yejin Choi, Yonatan Bisk, Yulia Tsvetkov","submitted_at":"2024-12-12T21:29:00Z","abstract_excerpt":"Do large language models (LLMs) have theory of mind? A plethora of papers and benchmarks have been introduced to evaluate if current models have been able to develop this key ability of social intelligence. However, all rely on limited datasets with simple patterns that can potentially lead to problematic blind spots in evaluation and an overestimation of model capabilities. We introduce ExploreToM, the first framework to allow large-scale generation of diverse and challenging theory of mind data for robust training and evaluation. Our approach leverages an A* search over a custom domain-speci"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.12175","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.12175/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.12175","created_at":"2026-07-05T09:50:11.079384+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.12175v1","created_at":"2026-07-05T09:50:11.079384+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.12175","created_at":"2026-07-05T09:50:11.079384+00:00"},{"alias_kind":"pith_short_12","alias_value":"WSG62FUGLCKN","created_at":"2026-07-05T09:50:11.079384+00:00"},{"alias_kind":"pith_short_16","alias_value":"WSG62FUGLCKNM2NF","created_at":"2026-07-05T09:50:11.079384+00:00"},{"alias_kind":"pith_short_8","alias_value":"WSG62FUG","created_at":"2026-07-05T09:50:11.079384+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00240","citing_title":"MindZero: Learning Online Mental Reasoning With Zero Annotations","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31916","citing_title":"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08415","citing_title":"Political Plasticity: An Analysis of Ideological Adaptability in Large Language Models","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WSG62FUGLCKNM2NF4SHLYNJPO3","json":"https://pith.science/pith/WSG62FUGLCKNM2NF4SHLYNJPO3.json","graph_json":"https://pith.science/api/pith-number/WSG62FUGLCKNM2NF4SHLYNJPO3/graph.json","events_json":"https://pith.science/api/pith-number/WSG62FUGLCKNM2NF4SHLYNJPO3/events.json","paper":"https://pith.science/paper/WSG62FUG"},"agent_actions":{"view_html":"https://pith.science/pith/WSG62FUGLCKNM2NF4SHLYNJPO3","download_json":"https://pith.science/pith/WSG62FUGLCKNM2NF4SHLYNJPO3.json","view_paper":"https://pith.science/paper/WSG62FUG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.12175&json=true","fetch_graph":"https://pith.science/api/pith-number/WSG62FUGLCKNM2NF4SHLYNJPO3/graph.json","fetch_events":"https://pith.science/api/pith-number/WSG62FUGLCKNM2NF4SHLYNJPO3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WSG62FUGLCKNM2NF4SHLYNJPO3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WSG62FUGLCKNM2NF4SHLYNJPO3/action/storage_attestation","attest_author":"https://pith.science/pith/WSG62FUGLCKNM2NF4SHLYNJPO3/action/author_attestation","sign_citation":"https://pith.science/pith/WSG62FUGLCKNM2NF4SHLYNJPO3/action/citation_signature","submit_replication":"https://pith.science/pith/WSG62FUGLCKNM2NF4SHLYNJPO3/action/replication_record"}},"created_at":"2026-07-05T09:50:11.079384+00:00","updated_at":"2026-07-05T09:50:11.079384+00:00"}