{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BD5ZO3FJ3RGJRIVK3QNKYEMYS2","short_pith_number":"pith:BD5ZO3FJ","schema_version":"1.0","canonical_sha256":"08fb976ca9dc4c98a2aadc1aac11989688279d0dee7467ffda23209c4731acf4","source":{"kind":"arxiv","id":"2406.19501","version":2},"attestation_state":"computed","paper":{"title":"Monitoring Latent World States in Language Models with Propositional Probes","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jacob Steinhardt, Jiahai Feng, Stuart Russell","submitted_at":"2024-06-27T19:28:43Z","abstract_excerpt":"Language models are susceptible to bias, sycophancy, backdoors, and other tendencies that lead to unfaithful responses to the input context. Interpreting internal states of language models could help monitor and correct unfaithful behavior. We hypothesize that language models represent their input contexts in a latent world model, and seek to extract this latent world state from the activations. We do so with 'propositional probes', which compositionally probe tokens for lexical information and bind them into logical propositions representing the world state. For example, given the input conte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.19501","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-27T19:28:43Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"7f84af1a9f6ac52965505dacb46cc063431634c7deae60d54f8233f3b030eb88","abstract_canon_sha256":"4637aba8d524e8ddce889383f55a7c2135d230c610305a5ecda90a2c28e7dc8f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:50.865554Z","signature_b64":"YFljCrIjGibutKt/2XK0kLGGKzXYbjy+S6/QpfyhqRbNR6azJvnjhZgQESKr5LkHx9fiPJ7eC3Md7aNTxdZjDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"08fb976ca9dc4c98a2aadc1aac11989688279d0dee7467ffda23209c4731acf4","last_reissued_at":"2026-07-05T09:45:50.865097Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:50.865097Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Monitoring Latent World States in Language Models with Propositional Probes","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jacob Steinhardt, Jiahai Feng, Stuart Russell","submitted_at":"2024-06-27T19:28:43Z","abstract_excerpt":"Language models are susceptible to bias, sycophancy, backdoors, and other tendencies that lead to unfaithful responses to the input context. Interpreting internal states of language models could help monitor and correct unfaithful behavior. We hypothesize that language models represent their input contexts in a latent world model, and seek to extract this latent world state from the activations. We do so with 'propositional probes', which compositionally probe tokens for lexical information and bind them into logical propositions representing the world state. For example, given the input conte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.19501","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.19501/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.19501","created_at":"2026-07-05T09:45:50.865157+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.19501v2","created_at":"2026-07-05T09:45:50.865157+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.19501","created_at":"2026-07-05T09:45:50.865157+00:00"},{"alias_kind":"pith_short_12","alias_value":"BD5ZO3FJ3RGJ","created_at":"2026-07-05T09:45:50.865157+00:00"},{"alias_kind":"pith_short_16","alias_value":"BD5ZO3FJ3RGJRIVK","created_at":"2026-07-05T09:45:50.865157+00:00"},{"alias_kind":"pith_short_8","alias_value":"BD5ZO3FJ","created_at":"2026-07-05T09:45:50.865157+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26523","citing_title":"Radical AI Interpretability","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18284","citing_title":"Breaking the Solver Bottleneck: Training Task Generators at the Learnable Frontier","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17408","citing_title":"The Impact of Off-Policy Training Data on Probe Generalisation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20090","citing_title":"Less Languages, Less Tokens: An Efficient Unified Logic Cross-lingual Chain-of-Thought Reasoning Framework","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19052","citing_title":"Cell-Based Representation of Relational Binding in Language Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15726","citing_title":"LLM Reasoning Is Latent, Not the Chain of Thought","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BD5ZO3FJ3RGJRIVK3QNKYEMYS2","json":"https://pith.science/pith/BD5ZO3FJ3RGJRIVK3QNKYEMYS2.json","graph_json":"https://pith.science/api/pith-number/BD5ZO3FJ3RGJRIVK3QNKYEMYS2/graph.json","events_json":"https://pith.science/api/pith-number/BD5ZO3FJ3RGJRIVK3QNKYEMYS2/events.json","paper":"https://pith.science/paper/BD5ZO3FJ"},"agent_actions":{"view_html":"https://pith.science/pith/BD5ZO3FJ3RGJRIVK3QNKYEMYS2","download_json":"https://pith.science/pith/BD5ZO3FJ3RGJRIVK3QNKYEMYS2.json","view_paper":"https://pith.science/paper/BD5ZO3FJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.19501&json=true","fetch_graph":"https://pith.science/api/pith-number/BD5ZO3FJ3RGJRIVK3QNKYEMYS2/graph.json","fetch_events":"https://pith.science/api/pith-number/BD5ZO3FJ3RGJRIVK3QNKYEMYS2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BD5ZO3FJ3RGJRIVK3QNKYEMYS2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BD5ZO3FJ3RGJRIVK3QNKYEMYS2/action/storage_attestation","attest_author":"https://pith.science/pith/BD5ZO3FJ3RGJRIVK3QNKYEMYS2/action/author_attestation","sign_citation":"https://pith.science/pith/BD5ZO3FJ3RGJRIVK3QNKYEMYS2/action/citation_signature","submit_replication":"https://pith.science/pith/BD5ZO3FJ3RGJRIVK3QNKYEMYS2/action/replication_record"}},"created_at":"2026-07-05T09:45:50.865157+00:00","updated_at":"2026-07-05T09:45:50.865157+00:00"}