{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:U7LMDM5BKJ2CODPHXU67S5DBUU","short_pith_number":"pith:U7LMDM5B","canonical_record":{"source":{"id":"2310.06253","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-10T01:58:38Z","cross_cats_sorted":[],"title_canon_sha256":"2798a72ebe1061f129127ebad195809d3d3cd324cc9c2e648f69bc01cda75fd8","abstract_canon_sha256":"63e96df75f79db4fd8241e9dbf07afe9c9f4c8ae7623606c90047b944bb617d2"},"schema_version":"1.0"},"canonical_sha256":"a7d6c1b3a15274270de7bd3df97461a500a6e1bfb77b738aabd223e3074241d5","source":{"kind":"arxiv","id":"2310.06253","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.06253","created_at":"2026-07-05T08:05:03Z"},{"alias_kind":"arxiv_version","alias_value":"2310.06253v2","created_at":"2026-07-05T08:05:03Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06253","created_at":"2026-07-05T08:05:03Z"},{"alias_kind":"pith_short_12","alias_value":"U7LMDM5BKJ2C","created_at":"2026-07-05T08:05:03Z"},{"alias_kind":"pith_short_16","alias_value":"U7LMDM5BKJ2CODPH","created_at":"2026-07-05T08:05:03Z"},{"alias_kind":"pith_short_8","alias_value":"U7LMDM5B","created_at":"2026-07-05T08:05:03Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:U7LMDM5BKJ2CODPHXU67S5DBUU","target":"record","payload":{"canonical_record":{"source":{"id":"2310.06253","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-10T01:58:38Z","cross_cats_sorted":[],"title_canon_sha256":"2798a72ebe1061f129127ebad195809d3d3cd324cc9c2e648f69bc01cda75fd8","abstract_canon_sha256":"63e96df75f79db4fd8241e9dbf07afe9c9f4c8ae7623606c90047b944bb617d2"},"schema_version":"1.0"},"canonical_sha256":"a7d6c1b3a15274270de7bd3df97461a500a6e1bfb77b738aabd223e3074241d5","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:05:03.652394Z","signature_b64":"rpNI9/hELsMFnv6G1mB6+sORY9IC8fsg8JqsWWBxvxwwzdX2yFHjLx1B2gdBLuhhe2pYRxSGCaqiQN77nGwUBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a7d6c1b3a15274270de7bd3df97461a500a6e1bfb77b738aabd223e3074241d5","last_reissued_at":"2026-07-05T08:05:03.651899Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:05:03.651899Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2310.06253","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:05:03Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"tsBR7vt4gWFA2yaX8UpwRI4vP5lIZJSkpoIEjfxB6ZwDsvPDWueH0VXWDmEhJI21iaG+Pl3pJVsWPWu84qX9Cg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T17:01:26.885840Z"},"content_sha256":"ebfba6c16361d8bd3ff26e946c49dc4a552105016730c03fd68da9d23eba8acc","schema_version":"1.0","event_id":"sha256:ebfba6c16361d8bd3ff26e946c49dc4a552105016730c03fd68da9d23eba8acc"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:U7LMDM5BKJ2CODPHXU67S5DBUU","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"A Unified View on Solving Objective Mismatch in Model-Based Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alfredo Garcia, Anthony McDonald, Nathan Lambert, Ran Wei, Roberto Calandra","submitted_at":"2023-10-10T01:58:38Z","abstract_excerpt":"Model-based Reinforcement Learning (MBRL) aims to make agents more sample-efficient, adaptive, and explainable by learning an explicit model of the environment. While the capabilities of MBRL agents have significantly improved in recent years, how to best learn the model is still an unresolved question. The majority of MBRL algorithms aim at training the model to make accurate predictions about the environment and subsequently using the model to determine the most rewarding actions. However, recent research has shown that model predictive accuracy is often not correlated with action quality, t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06253","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.06253/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:05:03Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"LEojBlJrz/k/JDzqypT7JcWeQ/O+RmpFw2wHSYjcISdc7+CosfrCkqDaDD4DJxbLAwGfQQ5XL/5MFEyekk4RDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T17:01:26.886464Z"},"content_sha256":"a2e387b3fa5ed44b2cc49cf9794c8ee2b5bb69a0c05554df1fa7b15f42479bbf","schema_version":"1.0","event_id":"sha256:a2e387b3fa5ed44b2cc49cf9794c8ee2b5bb69a0c05554df1fa7b15f42479bbf"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU/bundle.json","state_url":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/U7LMDM5BKJ2CODPHXU67S5DBUU/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T17:01:26Z","links":{"resolver":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU","bundle":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU/bundle.json","state":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU/state.json","well_known_bundle":"https://pith.science/.well-known/pith/U7LMDM5BKJ2CODPHXU67S5DBUU/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:U7LMDM5BKJ2CODPHXU67S5DBUU","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"63e96df75f79db4fd8241e9dbf07afe9c9f4c8ae7623606c90047b944bb617d2","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-10T01:58:38Z","title_canon_sha256":"2798a72ebe1061f129127ebad195809d3d3cd324cc9c2e648f69bc01cda75fd8"},"schema_version":"1.0","source":{"id":"2310.06253","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.06253","created_at":"2026-07-05T08:05:03Z"},{"alias_kind":"arxiv_version","alias_value":"2310.06253v2","created_at":"2026-07-05T08:05:03Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06253","created_at":"2026-07-05T08:05:03Z"},{"alias_kind":"pith_short_12","alias_value":"U7LMDM5BKJ2C","created_at":"2026-07-05T08:05:03Z"},{"alias_kind":"pith_short_16","alias_value":"U7LMDM5BKJ2CODPH","created_at":"2026-07-05T08:05:03Z"},{"alias_kind":"pith_short_8","alias_value":"U7LMDM5B","created_at":"2026-07-05T08:05:03Z"}],"graph_snapshots":[{"event_id":"sha256:a2e387b3fa5ed44b2cc49cf9794c8ee2b5bb69a0c05554df1fa7b15f42479bbf","target":"graph","created_at":"2026-07-05T08:05:03Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2310.06253/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Model-based Reinforcement Learning (MBRL) aims to make agents more sample-efficient, adaptive, and explainable by learning an explicit model of the environment. While the capabilities of MBRL agents have significantly improved in recent years, how to best learn the model is still an unresolved question. The majority of MBRL algorithms aim at training the model to make accurate predictions about the environment and subsequently using the model to determine the most rewarding actions. However, recent research has shown that model predictive accuracy is often not correlated with action quality, t","authors_text":"Alfredo Garcia, Anthony McDonald, Nathan Lambert, Ran Wei, Roberto Calandra","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-10T01:58:38Z","title":"A Unified View on Solving Objective Mismatch in Model-Based Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06253","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ebfba6c16361d8bd3ff26e946c49dc4a552105016730c03fd68da9d23eba8acc","target":"record","created_at":"2026-07-05T08:05:03Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"63e96df75f79db4fd8241e9dbf07afe9c9f4c8ae7623606c90047b944bb617d2","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-10T01:58:38Z","title_canon_sha256":"2798a72ebe1061f129127ebad195809d3d3cd324cc9c2e648f69bc01cda75fd8"},"schema_version":"1.0","source":{"id":"2310.06253","kind":"arxiv","version":2}},"canonical_sha256":"a7d6c1b3a15274270de7bd3df97461a500a6e1bfb77b738aabd223e3074241d5","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"a7d6c1b3a15274270de7bd3df97461a500a6e1bfb77b738aabd223e3074241d5","first_computed_at":"2026-07-05T08:05:03.651899Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:05:03.651899Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"rpNI9/hELsMFnv6G1mB6+sORY9IC8fsg8JqsWWBxvxwwzdX2yFHjLx1B2gdBLuhhe2pYRxSGCaqiQN77nGwUBg==","signature_status":"signed_v1","signed_at":"2026-07-05T08:05:03.652394Z","signed_message":"canonical_sha256_bytes"},"source_id":"2310.06253","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ebfba6c16361d8bd3ff26e946c49dc4a552105016730c03fd68da9d23eba8acc","sha256:a2e387b3fa5ed44b2cc49cf9794c8ee2b5bb69a0c05554df1fa7b15f42479bbf"],"state_sha256":"173c6e473c841136e7058d1ad289e7ca663670fd3b15097b79f78f6e1797763d"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4XJ5g/ktj+U0GqyPgrpS1LhNch8+nvGSzSY+k9G0q36K+evvp8Xw3YXtmsa+u6fsMQyj6UPYWqevSyLrFJc9AQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T17:01:26.890366Z","bundle_sha256":"ae929e0fc806cfb92f67ed0ce3c1acc965c4dab89c244181d15944947ec565a1"}}