{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:WN5ICM4KEVJ4OTO2VEEGHRKQTS","short_pith_number":"pith:WN5ICM4K","schema_version":"1.0","canonical_sha256":"b37a81338a2553c74ddaa90863c5509caa3cd20b1bfefdf779feb3242fdd1813","source":{"kind":"arxiv","id":"2602.12963","version":2},"attestation_state":"computed","paper":{"title":"Calculating Mutual Information between a Reward Maximizer and its Environment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alex Altair, Alfred Harwood, Jose Faustino","submitted_at":"2026-02-13T14:29:17Z","abstract_excerpt":"An important question in the field of AI is the extent to which successful behaviour requires an internal representation of the world. In this work, we quantify the amount of information an optimal policy provides about the underlying environment. We consider a Controlled Markov Process (CMP) with $n$ states and $m$ actions, assuming a uniform prior over the space of possible transition dynamics. We prove that observing a deterministic policy that is optimal for any non-constant reward function then conveys exactly $n \\log m$ bits of information about the environment. Specifically, we show tha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.12963","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-02-13T14:29:17Z","cross_cats_sorted":[],"title_canon_sha256":"70e339bb98231c91f27c3ebc23f448ddf89d6e164bcbf6a1d49de4112d4b1bc9","abstract_canon_sha256":"9ee1cf40e91034f5c1ee4979db30333239f5394739f8d18afc1e38aa2645d977"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-15T01:21:50.381526Z","signature_b64":"/CgeOLbtd+YtgMGkkgBUJPI+Vvr/jTFfTWiBJy3XcsETjPd1uPPAQGy54wurtVqW/0ZIAL1NdhtoksgWIijnCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b37a81338a2553c74ddaa90863c5509caa3cd20b1bfefdf779feb3242fdd1813","last_reissued_at":"2026-07-15T01:21:50.380706Z","signature_status":"signed_v1","first_computed_at":"2026-07-15T01:21:50.380706Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Calculating Mutual Information between a Reward Maximizer and its Environment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alex Altair, Alfred Harwood, Jose Faustino","submitted_at":"2026-02-13T14:29:17Z","abstract_excerpt":"An important question in the field of AI is the extent to which successful behaviour requires an internal representation of the world. In this work, we quantify the amount of information an optimal policy provides about the underlying environment. We consider a Controlled Markov Process (CMP) with $n$ states and $m$ actions, assuming a uniform prior over the space of possible transition dynamics. We prove that observing a deterministic policy that is optimal for any non-constant reward function then conveys exactly $n \\log m$ bits of information about the environment. Specifically, we show tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.12963","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.12963/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.12963","created_at":"2026-07-15T01:21:50.381119+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.12963v2","created_at":"2026-07-15T01:21:50.381119+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.12963","created_at":"2026-07-15T01:21:50.381119+00:00"},{"alias_kind":"pith_short_12","alias_value":"WN5ICM4KEVJ4","created_at":"2026-07-15T01:21:50.381119+00:00"},{"alias_kind":"pith_short_16","alias_value":"WN5ICM4KEVJ4OTO2","created_at":"2026-07-15T01:21:50.381119+00:00"},{"alias_kind":"pith_short_8","alias_value":"WN5ICM4K","created_at":"2026-07-15T01:21:50.381119+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WN5ICM4KEVJ4OTO2VEEGHRKQTS","json":"https://pith.science/pith/WN5ICM4KEVJ4OTO2VEEGHRKQTS.json","graph_json":"https://pith.science/api/pith-number/WN5ICM4KEVJ4OTO2VEEGHRKQTS/graph.json","events_json":"https://pith.science/api/pith-number/WN5ICM4KEVJ4OTO2VEEGHRKQTS/events.json","paper":"https://pith.science/paper/WN5ICM4K"},"agent_actions":{"view_html":"https://pith.science/pith/WN5ICM4KEVJ4OTO2VEEGHRKQTS","download_json":"https://pith.science/pith/WN5ICM4KEVJ4OTO2VEEGHRKQTS.json","view_paper":"https://pith.science/paper/WN5ICM4K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.12963&json=true","fetch_graph":"https://pith.science/api/pith-number/WN5ICM4KEVJ4OTO2VEEGHRKQTS/graph.json","fetch_events":"https://pith.science/api/pith-number/WN5ICM4KEVJ4OTO2VEEGHRKQTS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WN5ICM4KEVJ4OTO2VEEGHRKQTS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WN5ICM4KEVJ4OTO2VEEGHRKQTS/action/storage_attestation","attest_author":"https://pith.science/pith/WN5ICM4KEVJ4OTO2VEEGHRKQTS/action/author_attestation","sign_citation":"https://pith.science/pith/WN5ICM4KEVJ4OTO2VEEGHRKQTS/action/citation_signature","submit_replication":"https://pith.science/pith/WN5ICM4KEVJ4OTO2VEEGHRKQTS/action/replication_record"}},"created_at":"2026-07-15T01:21:50.381119+00:00","updated_at":"2026-07-15T01:21:50.381119+00:00"}