{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FYOHWGWM43PUWE3VA4VAABSVTG","short_pith_number":"pith:FYOHWGWM","schema_version":"1.0","canonical_sha256":"2e1c7b1acce6df4b1375072a00065599abeaa697ad235ba30e6efb929399021a","source":{"kind":"arxiv","id":"2507.22010","version":1},"attestation_state":"computed","paper":{"title":"Exploring the Stratified Space Structure of an RL Game with the Volume Growth Transform","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CG","cs.LG","math.DG"],"primary_cat":"math.AT","authors_text":"Alberto Speranzon, Brennan Lagasse, David Rosenbluth, Gregory Cox, Justin Curry, Ngoc B. Lam","submitted_at":"2025-07-29T17:00:33Z","abstract_excerpt":"In this work, we explore the structure of the embedding space of a transformer model trained for playing a particular reinforcement learning (RL) game. Specifically, we investigate how a transformer-based Proximal Policy Optimization (PPO) model embeds visual inputs in a simple environment where an agent must collect \"coins\" while avoiding dynamic obstacles consisting of \"spotlights.\" By adapting Robinson et al.'s study of the volume growth transform for LLMs to the RL setting, we find that the token embedding space for our visual coin collecting game is also not a manifold, and is better mode"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.22010","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.AT","submitted_at":"2025-07-29T17:00:33Z","cross_cats_sorted":["cs.AI","cs.CG","cs.LG","math.DG"],"title_canon_sha256":"1491ba5510c529fb87f0a0ac362f536609c1be9aac86c4243cc899d9b19ca6a3","abstract_canon_sha256":"27d163be027ae0141047b0e04c5c002cfe5b0381ff64eccd19354c700d6049df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:45:11.397987Z","signature_b64":"6VTNd8/VbP/EsWsqgBSOi2RkgcRzn0U21L+y9tEEVtxMeIrOnfvYXOYCKYk1qDaOVk4maPfEgCMEDIJ8Z/Q4CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2e1c7b1acce6df4b1375072a00065599abeaa697ad235ba30e6efb929399021a","last_reissued_at":"2026-07-05T11:45:11.397514Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:45:11.397514Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring the Stratified Space Structure of an RL Game with the Volume Growth Transform","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CG","cs.LG","math.DG"],"primary_cat":"math.AT","authors_text":"Alberto Speranzon, Brennan Lagasse, David Rosenbluth, Gregory Cox, Justin Curry, Ngoc B. Lam","submitted_at":"2025-07-29T17:00:33Z","abstract_excerpt":"In this work, we explore the structure of the embedding space of a transformer model trained for playing a particular reinforcement learning (RL) game. Specifically, we investigate how a transformer-based Proximal Policy Optimization (PPO) model embeds visual inputs in a simple environment where an agent must collect \"coins\" while avoiding dynamic obstacles consisting of \"spotlights.\" By adapting Robinson et al.'s study of the volume growth transform for LLMs to the RL setting, we find that the token embedding space for our visual coin collecting game is also not a manifold, and is better mode"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.22010","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.22010/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.22010","created_at":"2026-07-05T11:45:11.397572+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.22010v1","created_at":"2026-07-05T11:45:11.397572+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.22010","created_at":"2026-07-05T11:45:11.397572+00:00"},{"alias_kind":"pith_short_12","alias_value":"FYOHWGWM43PU","created_at":"2026-07-05T11:45:11.397572+00:00"},{"alias_kind":"pith_short_16","alias_value":"FYOHWGWM43PUWE3V","created_at":"2026-07-05T11:45:11.397572+00:00"},{"alias_kind":"pith_short_8","alias_value":"FYOHWGWM","created_at":"2026-07-05T11:45:11.397572+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.04923","citing_title":"Stratifying Reinforcement Learning with Signal Temporal Logic","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FYOHWGWM43PUWE3VA4VAABSVTG","json":"https://pith.science/pith/FYOHWGWM43PUWE3VA4VAABSVTG.json","graph_json":"https://pith.science/api/pith-number/FYOHWGWM43PUWE3VA4VAABSVTG/graph.json","events_json":"https://pith.science/api/pith-number/FYOHWGWM43PUWE3VA4VAABSVTG/events.json","paper":"https://pith.science/paper/FYOHWGWM"},"agent_actions":{"view_html":"https://pith.science/pith/FYOHWGWM43PUWE3VA4VAABSVTG","download_json":"https://pith.science/pith/FYOHWGWM43PUWE3VA4VAABSVTG.json","view_paper":"https://pith.science/paper/FYOHWGWM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.22010&json=true","fetch_graph":"https://pith.science/api/pith-number/FYOHWGWM43PUWE3VA4VAABSVTG/graph.json","fetch_events":"https://pith.science/api/pith-number/FYOHWGWM43PUWE3VA4VAABSVTG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FYOHWGWM43PUWE3VA4VAABSVTG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FYOHWGWM43PUWE3VA4VAABSVTG/action/storage_attestation","attest_author":"https://pith.science/pith/FYOHWGWM43PUWE3VA4VAABSVTG/action/author_attestation","sign_citation":"https://pith.science/pith/FYOHWGWM43PUWE3VA4VAABSVTG/action/citation_signature","submit_replication":"https://pith.science/pith/FYOHWGWM43PUWE3VA4VAABSVTG/action/replication_record"}},"created_at":"2026-07-05T11:45:11.397572+00:00","updated_at":"2026-07-05T11:45:11.397572+00:00"}