{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:DPNI3R4MBVBBTBRHQLQ55Y33LI","short_pith_number":"pith:DPNI3R4M","schema_version":"1.0","canonical_sha256":"1bda8dc78c0d4219862782e1dee37b5a39e4a22e51f7e11d0e8b2cdbad0b3ad8","source":{"kind":"arxiv","id":"2602.24121","version":2},"attestation_state":"computed","paper":{"title":"Online World Modeling Enables Real-World Inverse Reinforcement Learning from Observation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Bat Nemekhbold, Byron Boots, Harine Ravichandiran, Kevin Huang, Richard Ebock, Rohan Baijal, Sanghun Jung, Siyang Shen, Tyler Han","submitted_at":"2026-02-27T15:58:11Z","abstract_excerpt":"Current methods in robot learning are fundamentally bottlenecked by one or more of: hand-designed rewards, simulation modeling, or action supervision (e.g. teleoperation) each requiring significant domain expertise, engineering effort, and robot-operator labor. Towards eliminating these bottlenecks, this work pursues observational learning via Inverse Reinforcement Learning from Observation (IRLfO) in which only access to task observations (e.g. video) is assumed. Due to the challenging setting and limitations of RL methods, IRLfO has thus far remained impractical for real-world robot learning"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.24121","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2026-02-27T15:58:11Z","cross_cats_sorted":[],"title_canon_sha256":"16bd18b9b86091a2cd3cf2f25f15584d63abcf1da7be27a1c5c7c22570019432","abstract_canon_sha256":"57e95bc34df37b062d8c27e12677b2a03081ab2a332c2e3444ae90755a26453b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-23T01:12:02.447215Z","signature_b64":"s1bXi5gr63bhH/ee2K8VUINtfhk+osNtX1WLQdgqQhmYyUZU+UpeEB/yQk6A0wq7DU3UJpKuTpaUbGTGZzGJCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1bda8dc78c0d4219862782e1dee37b5a39e4a22e51f7e11d0e8b2cdbad0b3ad8","last_reissued_at":"2026-06-23T01:12:02.446618Z","signature_status":"signed_v1","first_computed_at":"2026-06-23T01:12:02.446618Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Online World Modeling Enables Real-World Inverse Reinforcement Learning from Observation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Bat Nemekhbold, Byron Boots, Harine Ravichandiran, Kevin Huang, Richard Ebock, Rohan Baijal, Sanghun Jung, Siyang Shen, Tyler Han","submitted_at":"2026-02-27T15:58:11Z","abstract_excerpt":"Current methods in robot learning are fundamentally bottlenecked by one or more of: hand-designed rewards, simulation modeling, or action supervision (e.g. teleoperation) each requiring significant domain expertise, engineering effort, and robot-operator labor. Towards eliminating these bottlenecks, this work pursues observational learning via Inverse Reinforcement Learning from Observation (IRLfO) in which only access to task observations (e.g. video) is assumed. Due to the challenging setting and limitations of RL methods, IRLfO has thus far remained impractical for real-world robot learning"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.24121","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.24121/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.24121","created_at":"2026-06-23T01:12:02.446691+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.24121v2","created_at":"2026-06-23T01:12:02.446691+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.24121","created_at":"2026-06-23T01:12:02.446691+00:00"},{"alias_kind":"pith_short_12","alias_value":"DPNI3R4MBVBB","created_at":"2026-06-23T01:12:02.446691+00:00"},{"alias_kind":"pith_short_16","alias_value":"DPNI3R4MBVBBTBRH","created_at":"2026-06-23T01:12:02.446691+00:00"},{"alias_kind":"pith_short_8","alias_value":"DPNI3R4M","created_at":"2026-06-23T01:12:02.446691+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DPNI3R4MBVBBTBRHQLQ55Y33LI","json":"https://pith.science/pith/DPNI3R4MBVBBTBRHQLQ55Y33LI.json","graph_json":"https://pith.science/api/pith-number/DPNI3R4MBVBBTBRHQLQ55Y33LI/graph.json","events_json":"https://pith.science/api/pith-number/DPNI3R4MBVBBTBRHQLQ55Y33LI/events.json","paper":"https://pith.science/paper/DPNI3R4M"},"agent_actions":{"view_html":"https://pith.science/pith/DPNI3R4MBVBBTBRHQLQ55Y33LI","download_json":"https://pith.science/pith/DPNI3R4MBVBBTBRHQLQ55Y33LI.json","view_paper":"https://pith.science/paper/DPNI3R4M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.24121&json=true","fetch_graph":"https://pith.science/api/pith-number/DPNI3R4MBVBBTBRHQLQ55Y33LI/graph.json","fetch_events":"https://pith.science/api/pith-number/DPNI3R4MBVBBTBRHQLQ55Y33LI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DPNI3R4MBVBBTBRHQLQ55Y33LI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DPNI3R4MBVBBTBRHQLQ55Y33LI/action/storage_attestation","attest_author":"https://pith.science/pith/DPNI3R4MBVBBTBRHQLQ55Y33LI/action/author_attestation","sign_citation":"https://pith.science/pith/DPNI3R4MBVBBTBRHQLQ55Y33LI/action/citation_signature","submit_replication":"https://pith.science/pith/DPNI3R4MBVBBTBRHQLQ55Y33LI/action/replication_record"}},"created_at":"2026-06-23T01:12:02.446691+00:00","updated_at":"2026-06-23T01:12:02.446691+00:00"}