{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P5WE7NHA7HIF4AA55VLZHUSI7C","short_pith_number":"pith:P5WE7NHA","schema_version":"1.0","canonical_sha256":"7f6c4fb4e0f9d05e001ded5793d248f891b47783b4431b9abe592eca409f1d3d","source":{"kind":"arxiv","id":"2409.14986","version":1},"attestation_state":"computed","paper":{"title":"Evaluating Theory of (an uncertain) Mind: Predicting the Uncertain Beliefs of Others in Conversation Forecasting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Anthony Sicilia, Malihe Alikhani","submitted_at":"2024-09-23T13:05:25Z","abstract_excerpt":"Typically, when evaluating Theory of Mind, we consider the beliefs of others to be binary: held or not held. But what if someone is unsure about their own beliefs? How can we quantify this uncertainty? We propose a new suite of tasks, challenging language models (LMs) to model the uncertainty of others in dialogue. We design these tasks around conversation forecasting, wherein an agent forecasts an unobserved outcome to a conversation. Uniquely, we view interlocutors themselves as forecasters, asking an LM to predict the uncertainty of the interlocutors (a probability). We experiment with re-s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.14986","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-23T13:05:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b2cb00e573c3a5000550764d47c2ffc00ad8084ced3da5cf7d7dd1731d4f6c2b","abstract_canon_sha256":"cd2f9a5d99cd0089d230e71a36794456cbdbbf4633ed15d1b30ce608d4c3aafb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:34.326345Z","signature_b64":"eXWuHsvz2Y4qeA8qSbMuvZgFPJawom/BHXtaXKcIONLu0SCF2f2S6uAzdBV5rrEAymYWYwBrJaht9FAH76wOAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f6c4fb4e0f9d05e001ded5793d248f891b47783b4431b9abe592eca409f1d3d","last_reissued_at":"2026-07-05T09:10:34.325857Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:34.325857Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Theory of (an uncertain) Mind: Predicting the Uncertain Beliefs of Others in Conversation Forecasting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Anthony Sicilia, Malihe Alikhani","submitted_at":"2024-09-23T13:05:25Z","abstract_excerpt":"Typically, when evaluating Theory of Mind, we consider the beliefs of others to be binary: held or not held. But what if someone is unsure about their own beliefs? How can we quantify this uncertainty? We propose a new suite of tasks, challenging language models (LMs) to model the uncertainty of others in dialogue. We design these tasks around conversation forecasting, wherein an agent forecasts an unobserved outcome to a conversation. Uniquely, we view interlocutors themselves as forecasters, asking an LM to predict the uncertainty of the interlocutors (a probability). We experiment with re-s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.14986","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.14986/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.14986","created_at":"2026-07-05T09:10:34.325917+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.14986v1","created_at":"2026-07-05T09:10:34.325917+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.14986","created_at":"2026-07-05T09:10:34.325917+00:00"},{"alias_kind":"pith_short_12","alias_value":"P5WE7NHA7HIF","created_at":"2026-07-05T09:10:34.325917+00:00"},{"alias_kind":"pith_short_16","alias_value":"P5WE7NHA7HIF4AA5","created_at":"2026-07-05T09:10:34.325917+00:00"},{"alias_kind":"pith_short_8","alias_value":"P5WE7NHA","created_at":"2026-07-05T09:10:34.325917+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.17348","citing_title":"Better Slow than Sorry: Introducing Positive Friction for Reliable Dialogue Systems","ref_index":58,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P5WE7NHA7HIF4AA55VLZHUSI7C","json":"https://pith.science/pith/P5WE7NHA7HIF4AA55VLZHUSI7C.json","graph_json":"https://pith.science/api/pith-number/P5WE7NHA7HIF4AA55VLZHUSI7C/graph.json","events_json":"https://pith.science/api/pith-number/P5WE7NHA7HIF4AA55VLZHUSI7C/events.json","paper":"https://pith.science/paper/P5WE7NHA"},"agent_actions":{"view_html":"https://pith.science/pith/P5WE7NHA7HIF4AA55VLZHUSI7C","download_json":"https://pith.science/pith/P5WE7NHA7HIF4AA55VLZHUSI7C.json","view_paper":"https://pith.science/paper/P5WE7NHA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.14986&json=true","fetch_graph":"https://pith.science/api/pith-number/P5WE7NHA7HIF4AA55VLZHUSI7C/graph.json","fetch_events":"https://pith.science/api/pith-number/P5WE7NHA7HIF4AA55VLZHUSI7C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P5WE7NHA7HIF4AA55VLZHUSI7C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P5WE7NHA7HIF4AA55VLZHUSI7C/action/storage_attestation","attest_author":"https://pith.science/pith/P5WE7NHA7HIF4AA55VLZHUSI7C/action/author_attestation","sign_citation":"https://pith.science/pith/P5WE7NHA7HIF4AA55VLZHUSI7C/action/citation_signature","submit_replication":"https://pith.science/pith/P5WE7NHA7HIF4AA55VLZHUSI7C/action/replication_record"}},"created_at":"2026-07-05T09:10:34.325917+00:00","updated_at":"2026-07-05T09:10:34.325917+00:00"}