{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TPD736KIT7FJLLRCMCYZPHFDHL","short_pith_number":"pith:TPD736KI","schema_version":"1.0","canonical_sha256":"9bc7fdf9489fca95ae2260b1979ca33ae41051f2ad922ff88bc3e19e6efddc9b","source":{"kind":"arxiv","id":"2503.00799","version":2},"attestation_state":"computed","paper":{"title":"On Generalization Across Environments In Multi-Objective Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jayden Teoh, Peter Vamplew, Pradeep Varakantham","submitted_at":"2025-03-02T08:50:14Z","abstract_excerpt":"Real-world sequential decision-making tasks often require balancing trade-offs between multiple conflicting objectives, making Multi-Objective Reinforcement Learning (MORL) an increasingly prominent field of research. Despite recent advances, existing MORL literature has narrowly focused on performance within static environments, neglecting the importance of generalizing across diverse settings. Conversely, existing research on generalization in RL has always assumed scalar rewards, overlooking the inherent multi-objectivity of real-world problems. Generalization in the multi-objective context"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.00799","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-02T08:50:14Z","cross_cats_sorted":[],"title_canon_sha256":"0289c6dc1f96728b5d748939f60edf891a92d85a236e8a43ce0779fcf8af4bd2","abstract_canon_sha256":"c343dbf4f3ca383cc8e9958796bbd484a7cb4b9dd4ced1e99ed380cdd5ac38c0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:30:03.907771Z","signature_b64":"IkSEzAqGFi0uY359GeCiUXCAt6ObPKwZNzwp7lTX8PeRnUOP83l3llXXrd7SFj6U3a2sHQ70GkES/McOyPIRBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9bc7fdf9489fca95ae2260b1979ca33ae41051f2ad922ff88bc3e19e6efddc9b","last_reissued_at":"2026-07-05T10:30:03.907287Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:30:03.907287Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Generalization Across Environments In Multi-Objective Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jayden Teoh, Peter Vamplew, Pradeep Varakantham","submitted_at":"2025-03-02T08:50:14Z","abstract_excerpt":"Real-world sequential decision-making tasks often require balancing trade-offs between multiple conflicting objectives, making Multi-Objective Reinforcement Learning (MORL) an increasingly prominent field of research. Despite recent advances, existing MORL literature has narrowly focused on performance within static environments, neglecting the importance of generalizing across diverse settings. Conversely, existing research on generalization in RL has always assumed scalar rewards, overlooking the inherent multi-objectivity of real-world problems. Generalization in the multi-objective context"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.00799","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.00799/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.00799","created_at":"2026-07-05T10:30:03.907347+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.00799v2","created_at":"2026-07-05T10:30:03.907347+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.00799","created_at":"2026-07-05T10:30:03.907347+00:00"},{"alias_kind":"pith_short_12","alias_value":"TPD736KIT7FJ","created_at":"2026-07-05T10:30:03.907347+00:00"},{"alias_kind":"pith_short_16","alias_value":"TPD736KIT7FJLLRC","created_at":"2026-07-05T10:30:03.907347+00:00"},{"alias_kind":"pith_short_8","alias_value":"TPD736KI","created_at":"2026-07-05T10:30:03.907347+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.02401","citing_title":"Towards Agents That Know When They Don't Know: Uncertainty as a Control Signal for Structured Reasoning","ref_index":53,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TPD736KIT7FJLLRCMCYZPHFDHL","json":"https://pith.science/pith/TPD736KIT7FJLLRCMCYZPHFDHL.json","graph_json":"https://pith.science/api/pith-number/TPD736KIT7FJLLRCMCYZPHFDHL/graph.json","events_json":"https://pith.science/api/pith-number/TPD736KIT7FJLLRCMCYZPHFDHL/events.json","paper":"https://pith.science/paper/TPD736KI"},"agent_actions":{"view_html":"https://pith.science/pith/TPD736KIT7FJLLRCMCYZPHFDHL","download_json":"https://pith.science/pith/TPD736KIT7FJLLRCMCYZPHFDHL.json","view_paper":"https://pith.science/paper/TPD736KI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.00799&json=true","fetch_graph":"https://pith.science/api/pith-number/TPD736KIT7FJLLRCMCYZPHFDHL/graph.json","fetch_events":"https://pith.science/api/pith-number/TPD736KIT7FJLLRCMCYZPHFDHL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TPD736KIT7FJLLRCMCYZPHFDHL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TPD736KIT7FJLLRCMCYZPHFDHL/action/storage_attestation","attest_author":"https://pith.science/pith/TPD736KIT7FJLLRCMCYZPHFDHL/action/author_attestation","sign_citation":"https://pith.science/pith/TPD736KIT7FJLLRCMCYZPHFDHL/action/citation_signature","submit_replication":"https://pith.science/pith/TPD736KIT7FJLLRCMCYZPHFDHL/action/replication_record"}},"created_at":"2026-07-05T10:30:03.907347+00:00","updated_at":"2026-07-05T10:30:03.907347+00:00"}