{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BPVAHGDUD3NGG2CPPJXAVSAZ44","short_pith_number":"pith:BPVAHGDU","schema_version":"1.0","canonical_sha256":"0bea0398741eda63684f7a6e0ac819e7312d7f3c602fdefc52e668fbbeacf64c","source":{"kind":"arxiv","id":"2406.17513","version":3},"attestation_state":"computed","paper":{"title":"Brittle Minds, Fixable Activations: Understanding Belief Representations in Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Andreas Bulling, Constantin Ruhdorfer, Lei Shi, Matteo Bortoletto","submitted_at":"2024-06-25T12:51:06Z","abstract_excerpt":"Despite growing interest in Theory of Mind (ToM) tasks for evaluating language models (LMs), little is known about how LMs internally represent mental states of self and others. Understanding these internal mechanisms is critical - not only to move beyond surface-level performance, but also for model alignment and safety, where subtle misattributions of mental states may go undetected in generated outputs. In this work, we present the first systematic investigation of belief representations in LMs by probing models across different scales, training regimens, and prompts - using control tasks t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.17513","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-25T12:51:06Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ae6cab2cde98598f2de5648e776b67457812f07b32de765d0b2947b4d379774b","abstract_canon_sha256":"bc43f937827046b1250fa46d3a4422f6e0ba9b580c92a9fcea59d0fcd44ba86d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:21.229001Z","signature_b64":"JMSdrzg78jzquVZEl+b7iHAZSM/RnCa8rgDB/2f1DEYEhWn1DoLt/2dMl1g63cTq0nU8mG0exeF1NRoRVJ+rDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0bea0398741eda63684f7a6e0ac819e7312d7f3c602fdefc52e668fbbeacf64c","last_reissued_at":"2026-07-05T11:05:21.228547Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:21.228547Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Brittle Minds, Fixable Activations: Understanding Belief Representations in Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Andreas Bulling, Constantin Ruhdorfer, Lei Shi, Matteo Bortoletto","submitted_at":"2024-06-25T12:51:06Z","abstract_excerpt":"Despite growing interest in Theory of Mind (ToM) tasks for evaluating language models (LMs), little is known about how LMs internally represent mental states of self and others. Understanding these internal mechanisms is critical - not only to move beyond surface-level performance, but also for model alignment and safety, where subtle misattributions of mental states may go undetected in generated outputs. In this work, we present the first systematic investigation of belief representations in LMs by probing models across different scales, training regimens, and prompts - using control tasks t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.17513","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.17513/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.17513","created_at":"2026-07-05T11:05:21.228605+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.17513v3","created_at":"2026-07-05T11:05:21.228605+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.17513","created_at":"2026-07-05T11:05:21.228605+00:00"},{"alias_kind":"pith_short_12","alias_value":"BPVAHGDUD3NG","created_at":"2026-07-05T11:05:21.228605+00:00"},{"alias_kind":"pith_short_16","alias_value":"BPVAHGDUD3NGG2CP","created_at":"2026-07-05T11:05:21.228605+00:00"},{"alias_kind":"pith_short_8","alias_value":"BPVAHGDU","created_at":"2026-07-05T11:05:21.228605+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.10298","citing_title":"On Emergent Social World Models -- Evidence for Functional Integration of Theory of Mind and Pragmatic Reasoning in Language Models","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BPVAHGDUD3NGG2CPPJXAVSAZ44","json":"https://pith.science/pith/BPVAHGDUD3NGG2CPPJXAVSAZ44.json","graph_json":"https://pith.science/api/pith-number/BPVAHGDUD3NGG2CPPJXAVSAZ44/graph.json","events_json":"https://pith.science/api/pith-number/BPVAHGDUD3NGG2CPPJXAVSAZ44/events.json","paper":"https://pith.science/paper/BPVAHGDU"},"agent_actions":{"view_html":"https://pith.science/pith/BPVAHGDUD3NGG2CPPJXAVSAZ44","download_json":"https://pith.science/pith/BPVAHGDUD3NGG2CPPJXAVSAZ44.json","view_paper":"https://pith.science/paper/BPVAHGDU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.17513&json=true","fetch_graph":"https://pith.science/api/pith-number/BPVAHGDUD3NGG2CPPJXAVSAZ44/graph.json","fetch_events":"https://pith.science/api/pith-number/BPVAHGDUD3NGG2CPPJXAVSAZ44/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BPVAHGDUD3NGG2CPPJXAVSAZ44/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BPVAHGDUD3NGG2CPPJXAVSAZ44/action/storage_attestation","attest_author":"https://pith.science/pith/BPVAHGDUD3NGG2CPPJXAVSAZ44/action/author_attestation","sign_citation":"https://pith.science/pith/BPVAHGDUD3NGG2CPPJXAVSAZ44/action/citation_signature","submit_replication":"https://pith.science/pith/BPVAHGDUD3NGG2CPPJXAVSAZ44/action/replication_record"}},"created_at":"2026-07-05T11:05:21.228605+00:00","updated_at":"2026-07-05T11:05:21.228605+00:00"}