{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:Z2YZPX5DVOSHPPJRBTDNYYB4IU","short_pith_number":"pith:Z2YZPX5D","schema_version":"1.0","canonical_sha256":"ceb197dfa3aba477bd310cc6dc603c453d1492b25fd4d757991f986fd1e097ef","source":{"kind":"arxiv","id":"1910.01723","version":3},"attestation_state":"computed","paper":{"title":"Using Logical Specifications of Objectives in Multi-Objective Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anand Balakrishnan, David Wingate, Jyotirmoy Deshmukh, Kolby Nottingham","submitted_at":"2019-10-03T21:16:04Z","abstract_excerpt":"It is notoriously difficult to control the behavior of reinforcement learning agents. Agents often learn to exploit the environment or reward signal and need to be retrained multiple times. The multi-objective reinforcement learning (MORL) framework separates a reward function into several objectives. An ideal MORL agent learns to generalize to novel combinations of objectives allowing for better control of an agent's behavior without requiring retraining. Many MORL approaches use a weight vector to parameterize the importance of each objective. However, this approach suffers from lack of expr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.01723","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-10-03T21:16:04Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"d8245827089cfb4669cb83db68291a9e4a7779b6da4ba1297e38acfef8d1cda1","abstract_canon_sha256":"4c37e3715891a60b45d9c53580fe51cd3556c57142e675b6d462a9a3ac543e56"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:11:22.590488Z","signature_b64":"RRtu85RdikRvuKgYJ/CuMsR1uN6EMy2AHZb0hvOTx5b4lRpuTkC39UkbIobRWAb9KltV14+IlZAYRd8wzpsDDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ceb197dfa3aba477bd310cc6dc603c453d1492b25fd4d757991f986fd1e097ef","last_reissued_at":"2026-07-05T03:11:22.590067Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:11:22.590067Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Using Logical Specifications of Objectives in Multi-Objective Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anand Balakrishnan, David Wingate, Jyotirmoy Deshmukh, Kolby Nottingham","submitted_at":"2019-10-03T21:16:04Z","abstract_excerpt":"It is notoriously difficult to control the behavior of reinforcement learning agents. Agents often learn to exploit the environment or reward signal and need to be retrained multiple times. The multi-objective reinforcement learning (MORL) framework separates a reward function into several objectives. An ideal MORL agent learns to generalize to novel combinations of objectives allowing for better control of an agent's behavior without requiring retraining. Many MORL approaches use a weight vector to parameterize the importance of each objective. However, this approach suffers from lack of expr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.01723","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.01723/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.01723","created_at":"2026-07-05T03:11:22.590125+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.01723v3","created_at":"2026-07-05T03:11:22.590125+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.01723","created_at":"2026-07-05T03:11:22.590125+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z2YZPX5DVOSH","created_at":"2026-07-05T03:11:22.590125+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z2YZPX5DVOSHPPJR","created_at":"2026-07-05T03:11:22.590125+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z2YZPX5D","created_at":"2026-07-05T03:11:22.590125+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z2YZPX5DVOSHPPJRBTDNYYB4IU","json":"https://pith.science/pith/Z2YZPX5DVOSHPPJRBTDNYYB4IU.json","graph_json":"https://pith.science/api/pith-number/Z2YZPX5DVOSHPPJRBTDNYYB4IU/graph.json","events_json":"https://pith.science/api/pith-number/Z2YZPX5DVOSHPPJRBTDNYYB4IU/events.json","paper":"https://pith.science/paper/Z2YZPX5D"},"agent_actions":{"view_html":"https://pith.science/pith/Z2YZPX5DVOSHPPJRBTDNYYB4IU","download_json":"https://pith.science/pith/Z2YZPX5DVOSHPPJRBTDNYYB4IU.json","view_paper":"https://pith.science/paper/Z2YZPX5D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.01723&json=true","fetch_graph":"https://pith.science/api/pith-number/Z2YZPX5DVOSHPPJRBTDNYYB4IU/graph.json","fetch_events":"https://pith.science/api/pith-number/Z2YZPX5DVOSHPPJRBTDNYYB4IU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z2YZPX5DVOSHPPJRBTDNYYB4IU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z2YZPX5DVOSHPPJRBTDNYYB4IU/action/storage_attestation","attest_author":"https://pith.science/pith/Z2YZPX5DVOSHPPJRBTDNYYB4IU/action/author_attestation","sign_citation":"https://pith.science/pith/Z2YZPX5DVOSHPPJRBTDNYYB4IU/action/citation_signature","submit_replication":"https://pith.science/pith/Z2YZPX5DVOSHPPJRBTDNYYB4IU/action/replication_record"}},"created_at":"2026-07-05T03:11:22.590125+00:00","updated_at":"2026-07-05T03:11:22.590125+00:00"}