{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7UQZDT6FD4IDWFTRZX25KBQY3Y","short_pith_number":"pith:7UQZDT6F","schema_version":"1.0","canonical_sha256":"fd2191cfc51f103b1671cdf5d50618de1119ef73144941d97ea99356a20b59bd","source":{"kind":"arxiv","id":"2506.09276","version":4},"attestation_state":"computed","paper":{"title":"Learning The Minimum Action Distance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Anders Jonsson, Joshua B. Evans, Lorenzo Steccanella, \\\"Ozg\\\"ur \\c{S}im\\c{s}ek","submitted_at":"2025-06-10T22:27:11Z","abstract_excerpt":"This paper presents a state representation framework for Markov decision processes (MDPs) that can be learned solely from state trajectories, requiring neither reward signals nor the actions executed by the agent. We propose learning the minimum action distance (MAD), defined as the minimum number of actions required to transition between states, as a fundamental metric that captures the underlying structure of an environment. MAD naturally enables critical downstream tasks such as goal-conditioned reinforcement learning and reward shaping by providing a dense, geometrically meaningful measure"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09276","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-10T22:27:11Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e6a9b7345bfaf0f266eb7babe223b9149c20ab27d842f0bfaf0943179460cc86","abstract_canon_sha256":"0b94e39ee9cae7951d058b7dc74089832d73366fe4233938becfca5e5bf0adb5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-08T01:18:19.016575Z","signature_b64":"AJA6E6TNYTi/xfQ/VBDsoJE4eGN+h2iNTwaWh08ysErQeM1OmD4rbG0UAd/adZfnziP1qfsoBbWchcVfwzGZBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fd2191cfc51f103b1671cdf5d50618de1119ef73144941d97ea99356a20b59bd","last_reissued_at":"2026-07-08T01:18:19.015979Z","signature_status":"signed_v1","first_computed_at":"2026-07-08T01:18:19.015979Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning The Minimum Action Distance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Anders Jonsson, Joshua B. Evans, Lorenzo Steccanella, \\\"Ozg\\\"ur \\c{S}im\\c{s}ek","submitted_at":"2025-06-10T22:27:11Z","abstract_excerpt":"This paper presents a state representation framework for Markov decision processes (MDPs) that can be learned solely from state trajectories, requiring neither reward signals nor the actions executed by the agent. We propose learning the minimum action distance (MAD), defined as the minimum number of actions required to transition between states, as a fundamental metric that captures the underlying structure of an environment. MAD naturally enables critical downstream tasks such as goal-conditioned reinforcement learning and reward shaping by providing a dense, geometrically meaningful measure"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09276","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09276/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09276","created_at":"2026-07-08T01:18:19.016045+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09276v4","created_at":"2026-07-08T01:18:19.016045+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09276","created_at":"2026-07-08T01:18:19.016045+00:00"},{"alias_kind":"pith_short_12","alias_value":"7UQZDT6FD4ID","created_at":"2026-07-08T01:18:19.016045+00:00"},{"alias_kind":"pith_short_16","alias_value":"7UQZDT6FD4IDWFTR","created_at":"2026-07-08T01:18:19.016045+00:00"},{"alias_kind":"pith_short_8","alias_value":"7UQZDT6F","created_at":"2026-07-08T01:18:19.016045+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7UQZDT6FD4IDWFTRZX25KBQY3Y","json":"https://pith.science/pith/7UQZDT6FD4IDWFTRZX25KBQY3Y.json","graph_json":"https://pith.science/api/pith-number/7UQZDT6FD4IDWFTRZX25KBQY3Y/graph.json","events_json":"https://pith.science/api/pith-number/7UQZDT6FD4IDWFTRZX25KBQY3Y/events.json","paper":"https://pith.science/paper/7UQZDT6F"},"agent_actions":{"view_html":"https://pith.science/pith/7UQZDT6FD4IDWFTRZX25KBQY3Y","download_json":"https://pith.science/pith/7UQZDT6FD4IDWFTRZX25KBQY3Y.json","view_paper":"https://pith.science/paper/7UQZDT6F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09276&json=true","fetch_graph":"https://pith.science/api/pith-number/7UQZDT6FD4IDWFTRZX25KBQY3Y/graph.json","fetch_events":"https://pith.science/api/pith-number/7UQZDT6FD4IDWFTRZX25KBQY3Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7UQZDT6FD4IDWFTRZX25KBQY3Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7UQZDT6FD4IDWFTRZX25KBQY3Y/action/storage_attestation","attest_author":"https://pith.science/pith/7UQZDT6FD4IDWFTRZX25KBQY3Y/action/author_attestation","sign_citation":"https://pith.science/pith/7UQZDT6FD4IDWFTRZX25KBQY3Y/action/citation_signature","submit_replication":"https://pith.science/pith/7UQZDT6FD4IDWFTRZX25KBQY3Y/action/replication_record"}},"created_at":"2026-07-08T01:18:19.016045+00:00","updated_at":"2026-07-08T01:18:19.016045+00:00"}