{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GSMLPOETK7MOWA5AZCNKVRAMQF","short_pith_number":"pith:GSMLPOET","schema_version":"1.0","canonical_sha256":"3498b7b89357d8eb03a0c89aaac40c81530dfae9fa39cf5dfb21e5f7cff9b4e1","source":{"kind":"arxiv","id":"2407.17466","version":1},"attestation_state":"computed","paper":{"title":"Traversing Pareto Optimal Policies: Provably Efficient Multi-Objective Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Boxiang Lyu, Dake Zhang, Rui Yang, Shuang Qiu, Tong Zhang","submitted_at":"2024-07-24T17:58:49Z","abstract_excerpt":"This paper investigates multi-objective reinforcement learning (MORL), which focuses on learning Pareto optimal policies in the presence of multiple reward functions. Despite MORL's significant empirical success, there is still a lack of satisfactory understanding of various MORL optimization targets and efficient learning algorithms. Our work offers a systematic analysis of several optimization targets to assess their abilities to find all Pareto optimal policies and controllability over learned policies by the preferences for different objectives. We then identify Tchebycheff scalarization a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.17466","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-24T17:58:49Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"292c12a1f7d9fb96a9e37fef6fbc97b919762772b43fc87f973383490b2ff559","abstract_canon_sha256":"0c4e63acf8ee01df6ec9924a5a7787267da0d048ec7cd9fa9d228c1cde611475"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:08.904804Z","signature_b64":"JqSb5+v/vbuLomAFvXlJbN2LMZ758CzmgshF7QB4PyXhGZlTbdn8IxFLEQVMnJbSvQJ8NZUoZ+jD0OXG1SjjCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3498b7b89357d8eb03a0c89aaac40c81530dfae9fa39cf5dfb21e5f7cff9b4e1","last_reissued_at":"2026-07-05T08:48:08.904410Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:08.904410Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Traversing Pareto Optimal Policies: Provably Efficient Multi-Objective Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Boxiang Lyu, Dake Zhang, Rui Yang, Shuang Qiu, Tong Zhang","submitted_at":"2024-07-24T17:58:49Z","abstract_excerpt":"This paper investigates multi-objective reinforcement learning (MORL), which focuses on learning Pareto optimal policies in the presence of multiple reward functions. Despite MORL's significant empirical success, there is still a lack of satisfactory understanding of various MORL optimization targets and efficient learning algorithms. Our work offers a systematic analysis of several optimization targets to assess their abilities to find all Pareto optimal policies and controllability over learned policies by the preferences for different objectives. We then identify Tchebycheff scalarization a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.17466","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.17466/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.17466","created_at":"2026-07-05T08:48:08.904474+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.17466v1","created_at":"2026-07-05T08:48:08.904474+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.17466","created_at":"2026-07-05T08:48:08.904474+00:00"},{"alias_kind":"pith_short_12","alias_value":"GSMLPOETK7MO","created_at":"2026-07-05T08:48:08.904474+00:00"},{"alias_kind":"pith_short_16","alias_value":"GSMLPOETK7MOWA5A","created_at":"2026-07-05T08:48:08.904474+00:00"},{"alias_kind":"pith_short_8","alias_value":"GSMLPOET","created_at":"2026-07-05T08:48:08.904474+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20619","citing_title":"SURF: Steering the Scalarization Weight to Uniformly Traverse the Pareto Front","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12771","citing_title":"Adaptive Smooth Tchebycheff Attention for Multi-Objective Policy Optimization","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03900","citing_title":"Contextual Multi-Objective Optimization: Rethinking Objectives in Frontier AI Systems","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24532","citing_title":"A Reward-Free Viewpoint on Multi-Objective Reinforcement Learning","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GSMLPOETK7MOWA5AZCNKVRAMQF","json":"https://pith.science/pith/GSMLPOETK7MOWA5AZCNKVRAMQF.json","graph_json":"https://pith.science/api/pith-number/GSMLPOETK7MOWA5AZCNKVRAMQF/graph.json","events_json":"https://pith.science/api/pith-number/GSMLPOETK7MOWA5AZCNKVRAMQF/events.json","paper":"https://pith.science/paper/GSMLPOET"},"agent_actions":{"view_html":"https://pith.science/pith/GSMLPOETK7MOWA5AZCNKVRAMQF","download_json":"https://pith.science/pith/GSMLPOETK7MOWA5AZCNKVRAMQF.json","view_paper":"https://pith.science/paper/GSMLPOET","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.17466&json=true","fetch_graph":"https://pith.science/api/pith-number/GSMLPOETK7MOWA5AZCNKVRAMQF/graph.json","fetch_events":"https://pith.science/api/pith-number/GSMLPOETK7MOWA5AZCNKVRAMQF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GSMLPOETK7MOWA5AZCNKVRAMQF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GSMLPOETK7MOWA5AZCNKVRAMQF/action/storage_attestation","attest_author":"https://pith.science/pith/GSMLPOETK7MOWA5AZCNKVRAMQF/action/author_attestation","sign_citation":"https://pith.science/pith/GSMLPOETK7MOWA5AZCNKVRAMQF/action/citation_signature","submit_replication":"https://pith.science/pith/GSMLPOETK7MOWA5AZCNKVRAMQF/action/replication_record"}},"created_at":"2026-07-05T08:48:08.904474+00:00","updated_at":"2026-07-05T08:48:08.904474+00:00"}