{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2020:ABN3JC3OXEC5RIXOMTOZ4DCY3W","short_pith_number":"pith:ABN3JC3O","canonical_record":{"source":{"id":"2003.00030","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-02-28T19:18:18Z","cross_cats_sorted":[],"title_canon_sha256":"528a6efac41871d3bbe3100932be871566ed85ba7a1d4f88d43cd9d15245cb87","abstract_canon_sha256":"4117ea7f0b8c44396ef7d4a2b6ea2af3dd23a1f9376b95c339c17e7839745efc"},"schema_version":"1.0"},"canonical_sha256":"005bb48b6eb905d8a2ee64dd9e0c58ddb23dc7644f28dd5c6fac246766e0e2af","source":{"kind":"arxiv","id":"2003.00030","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2003.00030","created_at":"2026-07-05T02:04:05Z"},{"alias_kind":"arxiv_version","alias_value":"2003.00030v2","created_at":"2026-07-05T02:04:05Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.00030","created_at":"2026-07-05T02:04:05Z"},{"alias_kind":"pith_short_12","alias_value":"ABN3JC3OXEC5","created_at":"2026-07-05T02:04:05Z"},{"alias_kind":"pith_short_16","alias_value":"ABN3JC3OXEC5RIXO","created_at":"2026-07-05T02:04:05Z"},{"alias_kind":"pith_short_8","alias_value":"ABN3JC3O","created_at":"2026-07-05T02:04:05Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2020:ABN3JC3OXEC5RIXOMTOZ4DCY3W","target":"record","payload":{"canonical_record":{"source":{"id":"2003.00030","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-02-28T19:18:18Z","cross_cats_sorted":[],"title_canon_sha256":"528a6efac41871d3bbe3100932be871566ed85ba7a1d4f88d43cd9d15245cb87","abstract_canon_sha256":"4117ea7f0b8c44396ef7d4a2b6ea2af3dd23a1f9376b95c339c17e7839745efc"},"schema_version":"1.0"},"canonical_sha256":"005bb48b6eb905d8a2ee64dd9e0c58ddb23dc7644f28dd5c6fac246766e0e2af","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:04:05.582033Z","signature_b64":"OpD3d3rctqanLcVaKrsY6zFe+7afvkTTIzUHN6k/TO7Tg1nzjdPHugilE8z9PTD+QvzrhTeQVHQBDZq2VoSGCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"005bb48b6eb905d8a2ee64dd9e0c58ddb23dc7644f28dd5c6fac246766e0e2af","last_reissued_at":"2026-07-05T02:04:05.581695Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:04:05.581695Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2003.00030","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:04:05Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"gIdB1UxZOPSOC6L3aZ894FJI4oFBk3oIODOn+sVxTRTjVNDnYytLGTJfjV0BC1mlyGUuHxYehb4fScgwy6sTCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T09:57:13.272757Z"},"content_sha256":"74fec858dd27d9f27beca39b51ccf0a5cd05a11470a6a93908477f487c3d3b2e","schema_version":"1.0","event_id":"sha256:74fec858dd27d9f27beca39b51ccf0a5cd05a11470a6a93908477f487c3d3b2e"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2020:ABN3JC3OXEC5RIXOMTOZ4DCY3W","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Policy-Aware Model Learning for Policy Gradient Methods","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Amir-massoud Farahmand, Mohammad Ghavamzadeh, Romina Abachi","submitted_at":"2020-02-28T19:18:18Z","abstract_excerpt":"This paper considers the problem of learning a model in model-based reinforcement learning (MBRL). We examine how the planning module of an MBRL algorithm uses the model, and propose that the model learning module should incorporate the way the planner is going to use the model. This is in contrast to conventional model learning approaches, such as those based on maximum likelihood estimate, that learn a predictive model of the environment without explicitly considering the interaction of the model and the planner. We focus on policy gradient type of planning algorithms and derive new loss fun"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.00030","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2003.00030/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:04:05Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"T662nFfbMyQ6I0EVVSiPBSxzm3+agkWBgt6dg9jXeMwLdO5ybpwj/Np4mdj71Jp9T8g7Er+5aLwLROUo6wunBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T09:57:13.273307Z"},"content_sha256":"b734841a6bac82c32d43663fcd2bf8ebd9d2505de8be87020f740d7d5fe30641","schema_version":"1.0","event_id":"sha256:b734841a6bac82c32d43663fcd2bf8ebd9d2505de8be87020f740d7d5fe30641"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/ABN3JC3OXEC5RIXOMTOZ4DCY3W/bundle.json","state_url":"https://pith.science/pith/ABN3JC3OXEC5RIXOMTOZ4DCY3W/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/ABN3JC3OXEC5RIXOMTOZ4DCY3W/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T09:57:13Z","links":{"resolver":"https://pith.science/pith/ABN3JC3OXEC5RIXOMTOZ4DCY3W","bundle":"https://pith.science/pith/ABN3JC3OXEC5RIXOMTOZ4DCY3W/bundle.json","state":"https://pith.science/pith/ABN3JC3OXEC5RIXOMTOZ4DCY3W/state.json","well_known_bundle":"https://pith.science/.well-known/pith/ABN3JC3OXEC5RIXOMTOZ4DCY3W/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:ABN3JC3OXEC5RIXOMTOZ4DCY3W","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4117ea7f0b8c44396ef7d4a2b6ea2af3dd23a1f9376b95c339c17e7839745efc","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-02-28T19:18:18Z","title_canon_sha256":"528a6efac41871d3bbe3100932be871566ed85ba7a1d4f88d43cd9d15245cb87"},"schema_version":"1.0","source":{"id":"2003.00030","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2003.00030","created_at":"2026-07-05T02:04:05Z"},{"alias_kind":"arxiv_version","alias_value":"2003.00030v2","created_at":"2026-07-05T02:04:05Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.00030","created_at":"2026-07-05T02:04:05Z"},{"alias_kind":"pith_short_12","alias_value":"ABN3JC3OXEC5","created_at":"2026-07-05T02:04:05Z"},{"alias_kind":"pith_short_16","alias_value":"ABN3JC3OXEC5RIXO","created_at":"2026-07-05T02:04:05Z"},{"alias_kind":"pith_short_8","alias_value":"ABN3JC3O","created_at":"2026-07-05T02:04:05Z"}],"graph_snapshots":[{"event_id":"sha256:b734841a6bac82c32d43663fcd2bf8ebd9d2505de8be87020f740d7d5fe30641","target":"graph","created_at":"2026-07-05T02:04:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2003.00030/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"This paper considers the problem of learning a model in model-based reinforcement learning (MBRL). We examine how the planning module of an MBRL algorithm uses the model, and propose that the model learning module should incorporate the way the planner is going to use the model. This is in contrast to conventional model learning approaches, such as those based on maximum likelihood estimate, that learn a predictive model of the environment without explicitly considering the interaction of the model and the planner. We focus on policy gradient type of planning algorithms and derive new loss fun","authors_text":"Amir-massoud Farahmand, Mohammad Ghavamzadeh, Romina Abachi","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-02-28T19:18:18Z","title":"Policy-Aware Model Learning for Policy Gradient Methods"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.00030","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:74fec858dd27d9f27beca39b51ccf0a5cd05a11470a6a93908477f487c3d3b2e","target":"record","created_at":"2026-07-05T02:04:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4117ea7f0b8c44396ef7d4a2b6ea2af3dd23a1f9376b95c339c17e7839745efc","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-02-28T19:18:18Z","title_canon_sha256":"528a6efac41871d3bbe3100932be871566ed85ba7a1d4f88d43cd9d15245cb87"},"schema_version":"1.0","source":{"id":"2003.00030","kind":"arxiv","version":2}},"canonical_sha256":"005bb48b6eb905d8a2ee64dd9e0c58ddb23dc7644f28dd5c6fac246766e0e2af","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"005bb48b6eb905d8a2ee64dd9e0c58ddb23dc7644f28dd5c6fac246766e0e2af","first_computed_at":"2026-07-05T02:04:05.581695Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:04:05.581695Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"OpD3d3rctqanLcVaKrsY6zFe+7afvkTTIzUHN6k/TO7Tg1nzjdPHugilE8z9PTD+QvzrhTeQVHQBDZq2VoSGCg==","signature_status":"signed_v1","signed_at":"2026-07-05T02:04:05.582033Z","signed_message":"canonical_sha256_bytes"},"source_id":"2003.00030","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:74fec858dd27d9f27beca39b51ccf0a5cd05a11470a6a93908477f487c3d3b2e","sha256:b734841a6bac82c32d43663fcd2bf8ebd9d2505de8be87020f740d7d5fe30641"],"state_sha256":"a2a1d40a94e6fd045e2a007cbd870fba4577c72ea9f8729b005770ddeb2c36fb"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dkH+LtC8gvsXPqlYytqQkfql4EIxieTIj1ZuJuc9lH/+URw/ANJneMRe/VRtx3huD7QMGJci7PFucSHDuIucDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T09:57:13.278048Z","bundle_sha256":"d1307674f29589a69812802e22cc73351442cd2b1c1f9c8295c69eb61f546047"}}