{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:U7LMDM5BKJ2CODPHXU67S5DBUU","short_pith_number":"pith:U7LMDM5B","schema_version":"1.0","canonical_sha256":"a7d6c1b3a15274270de7bd3df97461a500a6e1bfb77b738aabd223e3074241d5","source":{"kind":"arxiv","id":"2310.06253","version":2},"attestation_state":"computed","paper":{"title":"A Unified View on Solving Objective Mismatch in Model-Based Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alfredo Garcia, Anthony McDonald, Nathan Lambert, Ran Wei, Roberto Calandra","submitted_at":"2023-10-10T01:58:38Z","abstract_excerpt":"Model-based Reinforcement Learning (MBRL) aims to make agents more sample-efficient, adaptive, and explainable by learning an explicit model of the environment. While the capabilities of MBRL agents have significantly improved in recent years, how to best learn the model is still an unresolved question. The majority of MBRL algorithms aim at training the model to make accurate predictions about the environment and subsequently using the model to determine the most rewarding actions. However, recent research has shown that model predictive accuracy is often not correlated with action quality, t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.06253","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-10T01:58:38Z","cross_cats_sorted":[],"title_canon_sha256":"2798a72ebe1061f129127ebad195809d3d3cd324cc9c2e648f69bc01cda75fd8","abstract_canon_sha256":"63e96df75f79db4fd8241e9dbf07afe9c9f4c8ae7623606c90047b944bb617d2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:05:03.652394Z","signature_b64":"rpNI9/hELsMFnv6G1mB6+sORY9IC8fsg8JqsWWBxvxwwzdX2yFHjLx1B2gdBLuhhe2pYRxSGCaqiQN77nGwUBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a7d6c1b3a15274270de7bd3df97461a500a6e1bfb77b738aabd223e3074241d5","last_reissued_at":"2026-07-05T08:05:03.651899Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:05:03.651899Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Unified View on Solving Objective Mismatch in Model-Based Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alfredo Garcia, Anthony McDonald, Nathan Lambert, Ran Wei, Roberto Calandra","submitted_at":"2023-10-10T01:58:38Z","abstract_excerpt":"Model-based Reinforcement Learning (MBRL) aims to make agents more sample-efficient, adaptive, and explainable by learning an explicit model of the environment. While the capabilities of MBRL agents have significantly improved in recent years, how to best learn the model is still an unresolved question. The majority of MBRL algorithms aim at training the model to make accurate predictions about the environment and subsequently using the model to determine the most rewarding actions. However, recent research has shown that model predictive accuracy is often not correlated with action quality, t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06253","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.06253/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.06253","created_at":"2026-07-05T08:05:03.651953+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.06253v2","created_at":"2026-07-05T08:05:03.651953+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06253","created_at":"2026-07-05T08:05:03.651953+00:00"},{"alias_kind":"pith_short_12","alias_value":"U7LMDM5BKJ2C","created_at":"2026-07-05T08:05:03.651953+00:00"},{"alias_kind":"pith_short_16","alias_value":"U7LMDM5BKJ2CODPH","created_at":"2026-07-05T08:05:03.651953+00:00"},{"alias_kind":"pith_short_8","alias_value":"U7LMDM5B","created_at":"2026-07-05T08:05:03.651953+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15032","citing_title":"How Should World Models Be Evaluated for Embodied Decision-Making? A Decision-Making-Centric Position","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU","json":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU.json","graph_json":"https://pith.science/api/pith-number/U7LMDM5BKJ2CODPHXU67S5DBUU/graph.json","events_json":"https://pith.science/api/pith-number/U7LMDM5BKJ2CODPHXU67S5DBUU/events.json","paper":"https://pith.science/paper/U7LMDM5B"},"agent_actions":{"view_html":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU","download_json":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU.json","view_paper":"https://pith.science/paper/U7LMDM5B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.06253&json=true","fetch_graph":"https://pith.science/api/pith-number/U7LMDM5BKJ2CODPHXU67S5DBUU/graph.json","fetch_events":"https://pith.science/api/pith-number/U7LMDM5BKJ2CODPHXU67S5DBUU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU/action/storage_attestation","attest_author":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU/action/author_attestation","sign_citation":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU/action/citation_signature","submit_replication":"https://pith.science/pith/U7LMDM5BKJ2CODPHXU67S5DBUU/action/replication_record"}},"created_at":"2026-07-05T08:05:03.651953+00:00","updated_at":"2026-07-05T08:05:03.651953+00:00"}