{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NNOWSPE25RSNSAP6BRGIGQDCIL","short_pith_number":"pith:NNOWSPE2","schema_version":"1.0","canonical_sha256":"6b5d693c9aec64d901fe0c4c83406242cc50916dc6068ddc03a68006151692d0","source":{"kind":"arxiv","id":"2304.01203","version":7},"attestation_state":"computed","paper":{"title":"Optimal Goal-Reaching Reinforcement Learning via Quasimetric Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amy Zhang, Antonio Torralba, Phillip Isola, Tongzhou Wang","submitted_at":"2023-04-03T17:59:58Z","abstract_excerpt":"In goal-reaching reinforcement learning (RL), the optimal value function has a particular geometry, called quasimetric structure. This paper introduces Quasimetric Reinforcement Learning (QRL), a new RL method that utilizes quasimetric models to learn optimal value functions. Distinct from prior approaches, the QRL objective is specifically designed for quasimetrics, and provides strong theoretical recovery guarantees. Empirically, we conduct thorough analyses on a discretized MountainCar environment, identifying properties of QRL and its advantages over alternatives. On offline and online goa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.01203","kind":"arxiv","version":7},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-04-03T17:59:58Z","cross_cats_sorted":[],"title_canon_sha256":"4e4e413da54fed85c6841899857e35f44128704836cd27c8e7054e7ca25ad976","abstract_canon_sha256":"f845ee7da6d0810566990e47a84d526c57a9fc4253fab018707e775dc0ad9394"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:16:39.629451Z","signature_b64":"X3I+Hq/tUckEud3uhLmNlKY/I+f98EeRxe2xL6RWUDJbSHPdkfvCQoXRXqFDIQ5e6z9EwleMKHZ7wPcKlweGCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6b5d693c9aec64d901fe0c4c83406242cc50916dc6068ddc03a68006151692d0","last_reissued_at":"2026-07-05T07:16:39.628955Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:16:39.628955Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimal Goal-Reaching Reinforcement Learning via Quasimetric Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amy Zhang, Antonio Torralba, Phillip Isola, Tongzhou Wang","submitted_at":"2023-04-03T17:59:58Z","abstract_excerpt":"In goal-reaching reinforcement learning (RL), the optimal value function has a particular geometry, called quasimetric structure. This paper introduces Quasimetric Reinforcement Learning (QRL), a new RL method that utilizes quasimetric models to learn optimal value functions. Distinct from prior approaches, the QRL objective is specifically designed for quasimetrics, and provides strong theoretical recovery guarantees. Empirically, we conduct thorough analyses on a discretized MountainCar environment, identifying properties of QRL and its advantages over alternatives. On offline and online goa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.01203","kind":"arxiv","version":7},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.01203/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.01203","created_at":"2026-07-05T07:16:39.629013+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.01203v7","created_at":"2026-07-05T07:16:39.629013+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.01203","created_at":"2026-07-05T07:16:39.629013+00:00"},{"alias_kind":"pith_short_12","alias_value":"NNOWSPE25RSN","created_at":"2026-07-05T07:16:39.629013+00:00"},{"alias_kind":"pith_short_16","alias_value":"NNOWSPE25RSNSAP6","created_at":"2026-07-05T07:16:39.629013+00:00"},{"alias_kind":"pith_short_8","alias_value":"NNOWSPE2","created_at":"2026-07-05T07:16:39.629013+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.10137","citing_title":"Self-Predictive Representations for Combinatorial Generalization in Behavioral Cloning","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NNOWSPE25RSNSAP6BRGIGQDCIL","json":"https://pith.science/pith/NNOWSPE25RSNSAP6BRGIGQDCIL.json","graph_json":"https://pith.science/api/pith-number/NNOWSPE25RSNSAP6BRGIGQDCIL/graph.json","events_json":"https://pith.science/api/pith-number/NNOWSPE25RSNSAP6BRGIGQDCIL/events.json","paper":"https://pith.science/paper/NNOWSPE2"},"agent_actions":{"view_html":"https://pith.science/pith/NNOWSPE25RSNSAP6BRGIGQDCIL","download_json":"https://pith.science/pith/NNOWSPE25RSNSAP6BRGIGQDCIL.json","view_paper":"https://pith.science/paper/NNOWSPE2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.01203&json=true","fetch_graph":"https://pith.science/api/pith-number/NNOWSPE25RSNSAP6BRGIGQDCIL/graph.json","fetch_events":"https://pith.science/api/pith-number/NNOWSPE25RSNSAP6BRGIGQDCIL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NNOWSPE25RSNSAP6BRGIGQDCIL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NNOWSPE25RSNSAP6BRGIGQDCIL/action/storage_attestation","attest_author":"https://pith.science/pith/NNOWSPE25RSNSAP6BRGIGQDCIL/action/author_attestation","sign_citation":"https://pith.science/pith/NNOWSPE25RSNSAP6BRGIGQDCIL/action/citation_signature","submit_replication":"https://pith.science/pith/NNOWSPE25RSNSAP6BRGIGQDCIL/action/replication_record"}},"created_at":"2026-07-05T07:16:39.629013+00:00","updated_at":"2026-07-05T07:16:39.629013+00:00"}