{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3TVDR5GG57FO6I6VANH7DVY6PY","short_pith_number":"pith:3TVDR5GG","schema_version":"1.0","canonical_sha256":"dcea38f4c6efcaef23d5034ff1d71e7e3635e2b971299b8387e2616f3986e3e2","source":{"kind":"arxiv","id":"2504.04608","version":1},"attestation_state":"computed","paper":{"title":"AI in a vat: Fundamental limits of efficient world modelling for agent sandboxing and interpretability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.AI","authors_text":"Alexander Boyd, Fernando Rosas, Manuel Baltieri","submitted_at":"2025-04-06T20:35:44Z","abstract_excerpt":"Recent work proposes using world models to generate controlled virtual environments in which AI agents can be tested before deployment to ensure their reliability and safety. However, accurate world models often have high computational demands that can severely restrict the scope and depth of such assessments. Inspired by the classic `brain in a vat' thought experiment, here we investigate ways of simplifying world models that remain agnostic to the AI agent under evaluation. By following principles from computational mechanics, our approach reveals a fundamental trade-off in world model const"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.04608","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-04-06T20:35:44Z","cross_cats_sorted":["cs.SY","eess.SY"],"title_canon_sha256":"1a3b5f5ea69d2d3c177abb5095d8638c7edb3ff6008dd905f329a2338d20189a","abstract_canon_sha256":"04f7e911aa6f8ead14ff1253c23b1aa186915ba2c26907919a4261e0c963fd47"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:26.948472Z","signature_b64":"BsBkoTbheOs7llZBQSqsh0ZP+/AF+RGodBCrbOL6g60EWQ1SQeaZ+XZ2cUPOWQIh3OaRAQrmG0QN7urYN+T1Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dcea38f4c6efcaef23d5034ff1d71e7e3635e2b971299b8387e2616f3986e3e2","last_reissued_at":"2026-07-05T10:45:26.947993Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:26.947993Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AI in a vat: Fundamental limits of efficient world modelling for agent sandboxing and interpretability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.AI","authors_text":"Alexander Boyd, Fernando Rosas, Manuel Baltieri","submitted_at":"2025-04-06T20:35:44Z","abstract_excerpt":"Recent work proposes using world models to generate controlled virtual environments in which AI agents can be tested before deployment to ensure their reliability and safety. However, accurate world models often have high computational demands that can severely restrict the scope and depth of such assessments. Inspired by the classic `brain in a vat' thought experiment, here we investigate ways of simplifying world models that remain agnostic to the AI agent under evaluation. By following principles from computational mechanics, our approach reveals a fundamental trade-off in world model const"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.04608","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.04608/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.04608","created_at":"2026-07-05T10:45:26.948049+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.04608v1","created_at":"2026-07-05T10:45:26.948049+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.04608","created_at":"2026-07-05T10:45:26.948049+00:00"},{"alias_kind":"pith_short_12","alias_value":"3TVDR5GG57FO","created_at":"2026-07-05T10:45:26.948049+00:00"},{"alias_kind":"pith_short_16","alias_value":"3TVDR5GG57FO6I6V","created_at":"2026-07-05T10:45:26.948049+00:00"},{"alias_kind":"pith_short_8","alias_value":"3TVDR5GG","created_at":"2026-07-05T10:45:26.948049+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.03004","citing_title":"Identifiability and minimality bounds of quantum and post-quantum models of classical stochastic processes","ref_index":41,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3TVDR5GG57FO6I6VANH7DVY6PY","json":"https://pith.science/pith/3TVDR5GG57FO6I6VANH7DVY6PY.json","graph_json":"https://pith.science/api/pith-number/3TVDR5GG57FO6I6VANH7DVY6PY/graph.json","events_json":"https://pith.science/api/pith-number/3TVDR5GG57FO6I6VANH7DVY6PY/events.json","paper":"https://pith.science/paper/3TVDR5GG"},"agent_actions":{"view_html":"https://pith.science/pith/3TVDR5GG57FO6I6VANH7DVY6PY","download_json":"https://pith.science/pith/3TVDR5GG57FO6I6VANH7DVY6PY.json","view_paper":"https://pith.science/paper/3TVDR5GG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.04608&json=true","fetch_graph":"https://pith.science/api/pith-number/3TVDR5GG57FO6I6VANH7DVY6PY/graph.json","fetch_events":"https://pith.science/api/pith-number/3TVDR5GG57FO6I6VANH7DVY6PY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3TVDR5GG57FO6I6VANH7DVY6PY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3TVDR5GG57FO6I6VANH7DVY6PY/action/storage_attestation","attest_author":"https://pith.science/pith/3TVDR5GG57FO6I6VANH7DVY6PY/action/author_attestation","sign_citation":"https://pith.science/pith/3TVDR5GG57FO6I6VANH7DVY6PY/action/citation_signature","submit_replication":"https://pith.science/pith/3TVDR5GG57FO6I6VANH7DVY6PY/action/replication_record"}},"created_at":"2026-07-05T10:45:26.948049+00:00","updated_at":"2026-07-05T10:45:26.948049+00:00"}