{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:GB2B3GSMRS3H4MUU6KPAQY4PUN","short_pith_number":"pith:GB2B3GSM","schema_version":"1.0","canonical_sha256":"30741d9a4c8cb67e3294f29e08638fa34368ce32f0eb584e0ba3ef6cfbc90cb7","source":{"kind":"arxiv","id":"2607.14180","version":1},"attestation_state":"computed","paper":{"title":"RENEW: Towards Learning World Models and Repairing Model Exploitation from Preferences","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Logan Mondal Bhamidipaty, Mykel Kochenderfer, Subramanian Ramamoorthy","submitted_at":"2026-07-15T14:03:45Z","abstract_excerpt":"World models are widely used in offline reinforcement learning (RL) to improve sample efficiency and generate experience beyond a fixed dataset. However, they are vulnerable to model exploitation where data coverage is thin. Prior work addresses this either by collecting more expert demonstrations, which is often expensive, unsafe, or unavailable, or by conservative algorithms that avoid uncertain regions, which limits generalization. We propose instead to repair exploitation directly using human preferences over imagined rollouts, leveraging the strong intuitive physics that allows humans to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.14180","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-15T14:03:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"09c43732a273b5f3e92472eb5ce2509797d9bf0a6c43c0f82d2159a5de7c77ae","abstract_canon_sha256":"855d6e58236986b2f5906214119b6d89097cc6279a1c6418e8b3620d006cbafe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-17T00:20:55.695920Z","signature_b64":"Ngvd9s562dQUIyMvvwGDc5O6rnK72ZZuOdZ8d3TDx2WM4b7J+cW5L2nnFHE6fdieR/k9oDPV/YwQ5Mr0U0g3Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"30741d9a4c8cb67e3294f29e08638fa34368ce32f0eb584e0ba3ef6cfbc90cb7","last_reissued_at":"2026-07-17T00:20:55.695003Z","signature_status":"signed_v1","first_computed_at":"2026-07-17T00:20:55.695003Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RENEW: Towards Learning World Models and Repairing Model Exploitation from Preferences","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Logan Mondal Bhamidipaty, Mykel Kochenderfer, Subramanian Ramamoorthy","submitted_at":"2026-07-15T14:03:45Z","abstract_excerpt":"World models are widely used in offline reinforcement learning (RL) to improve sample efficiency and generate experience beyond a fixed dataset. However, they are vulnerable to model exploitation where data coverage is thin. Prior work addresses this either by collecting more expert demonstrations, which is often expensive, unsafe, or unavailable, or by conservative algorithms that avoid uncertain regions, which limits generalization. We propose instead to repair exploitation directly using human preferences over imagined rollouts, leveraging the strong intuitive physics that allows humans to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.14180","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.14180/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.14180","created_at":"2026-07-17T00:20:55.695482+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.14180v1","created_at":"2026-07-17T00:20:55.695482+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.14180","created_at":"2026-07-17T00:20:55.695482+00:00"},{"alias_kind":"pith_short_12","alias_value":"GB2B3GSMRS3H","created_at":"2026-07-17T00:20:55.695482+00:00"},{"alias_kind":"pith_short_16","alias_value":"GB2B3GSMRS3H4MUU","created_at":"2026-07-17T00:20:55.695482+00:00"},{"alias_kind":"pith_short_8","alias_value":"GB2B3GSM","created_at":"2026-07-17T00:20:55.695482+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GB2B3GSMRS3H4MUU6KPAQY4PUN","json":"https://pith.science/pith/GB2B3GSMRS3H4MUU6KPAQY4PUN.json","graph_json":"https://pith.science/api/pith-number/GB2B3GSMRS3H4MUU6KPAQY4PUN/graph.json","events_json":"https://pith.science/api/pith-number/GB2B3GSMRS3H4MUU6KPAQY4PUN/events.json","paper":"https://pith.science/paper/GB2B3GSM"},"agent_actions":{"view_html":"https://pith.science/pith/GB2B3GSMRS3H4MUU6KPAQY4PUN","download_json":"https://pith.science/pith/GB2B3GSMRS3H4MUU6KPAQY4PUN.json","view_paper":"https://pith.science/paper/GB2B3GSM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.14180&json=true","fetch_graph":"https://pith.science/api/pith-number/GB2B3GSMRS3H4MUU6KPAQY4PUN/graph.json","fetch_events":"https://pith.science/api/pith-number/GB2B3GSMRS3H4MUU6KPAQY4PUN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GB2B3GSMRS3H4MUU6KPAQY4PUN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GB2B3GSMRS3H4MUU6KPAQY4PUN/action/storage_attestation","attest_author":"https://pith.science/pith/GB2B3GSMRS3H4MUU6KPAQY4PUN/action/author_attestation","sign_citation":"https://pith.science/pith/GB2B3GSMRS3H4MUU6KPAQY4PUN/action/citation_signature","submit_replication":"https://pith.science/pith/GB2B3GSMRS3H4MUU6KPAQY4PUN/action/replication_record"}},"created_at":"2026-07-17T00:20:55.695482+00:00","updated_at":"2026-07-17T00:20:55.695482+00:00"}