{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TYCWQCWPSFJBBFUP5LOHMG47EL","short_pith_number":"pith:TYCWQCWP","schema_version":"1.0","canonical_sha256":"9e05680acf915210968feadc761b9f22d462722dd4773153a21cdef8aeddf04a","source":{"kind":"arxiv","id":"2304.04782","version":1},"attestation_state":"computed","paper":{"title":"Reinforcement Learning from Passive Data via Latent Intentions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chethan Bhateja, Dibya Ghosh, Sergey Levine","submitted_at":"2023-04-10T17:59:05Z","abstract_excerpt":"Passive observational data, such as human videos, is abundant and rich in information, yet remains largely untapped by current RL methods. Perhaps surprisingly, we show that passive data, despite not having reward or action labels, can still be used to learn features that accelerate downstream RL. Our approach learns from passive data by modeling intentions: measuring how the likelihood of future outcomes change when the agent acts to achieve a particular task. We propose a temporal difference learning objective to learn about intentions, resulting in an algorithm similar to conventional RL, b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.04782","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-04-10T17:59:05Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"e67899fb20054c7aae685b393071858d1175b6994eee662ad1547ccd80a72618","abstract_canon_sha256":"3cd3e4afb45f5c2160808db62d53fa9b4855a33cefed08e1d5bab08e2102320a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:59:51.750671Z","signature_b64":"ts/VCb9QM8adxUPYBYBqdYR5JKSMWEieeIu8nUoAy4nVCgzEeMSCusjrlRMm8DcjlKj4ojCcfdOVQ4eSJQIsAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e05680acf915210968feadc761b9f22d462722dd4773153a21cdef8aeddf04a","last_reissued_at":"2026-07-05T05:59:51.750233Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:59:51.750233Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning from Passive Data via Latent Intentions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chethan Bhateja, Dibya Ghosh, Sergey Levine","submitted_at":"2023-04-10T17:59:05Z","abstract_excerpt":"Passive observational data, such as human videos, is abundant and rich in information, yet remains largely untapped by current RL methods. Perhaps surprisingly, we show that passive data, despite not having reward or action labels, can still be used to learn features that accelerate downstream RL. Our approach learns from passive data by modeling intentions: measuring how the likelihood of future outcomes change when the agent acts to achieve a particular task. We propose a temporal difference learning objective to learn about intentions, resulting in an algorithm similar to conventional RL, b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.04782","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.04782/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.04782","created_at":"2026-07-05T05:59:51.750289+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.04782v1","created_at":"2026-07-05T05:59:51.750289+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.04782","created_at":"2026-07-05T05:59:51.750289+00:00"},{"alias_kind":"pith_short_12","alias_value":"TYCWQCWPSFJB","created_at":"2026-07-05T05:59:51.750289+00:00"},{"alias_kind":"pith_short_16","alias_value":"TYCWQCWPSFJBBFUP","created_at":"2026-07-05T05:59:51.750289+00:00"},{"alias_kind":"pith_short_8","alias_value":"TYCWQCWP","created_at":"2026-07-05T05:59:51.750289+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22711","citing_title":"Abstraction for Offline Goal-Conditioned Reinforcement Learning","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TYCWQCWPSFJBBFUP5LOHMG47EL","json":"https://pith.science/pith/TYCWQCWPSFJBBFUP5LOHMG47EL.json","graph_json":"https://pith.science/api/pith-number/TYCWQCWPSFJBBFUP5LOHMG47EL/graph.json","events_json":"https://pith.science/api/pith-number/TYCWQCWPSFJBBFUP5LOHMG47EL/events.json","paper":"https://pith.science/paper/TYCWQCWP"},"agent_actions":{"view_html":"https://pith.science/pith/TYCWQCWPSFJBBFUP5LOHMG47EL","download_json":"https://pith.science/pith/TYCWQCWPSFJBBFUP5LOHMG47EL.json","view_paper":"https://pith.science/paper/TYCWQCWP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.04782&json=true","fetch_graph":"https://pith.science/api/pith-number/TYCWQCWPSFJBBFUP5LOHMG47EL/graph.json","fetch_events":"https://pith.science/api/pith-number/TYCWQCWPSFJBBFUP5LOHMG47EL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TYCWQCWPSFJBBFUP5LOHMG47EL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TYCWQCWPSFJBBFUP5LOHMG47EL/action/storage_attestation","attest_author":"https://pith.science/pith/TYCWQCWPSFJBBFUP5LOHMG47EL/action/author_attestation","sign_citation":"https://pith.science/pith/TYCWQCWPSFJBBFUP5LOHMG47EL/action/citation_signature","submit_replication":"https://pith.science/pith/TYCWQCWPSFJBBFUP5LOHMG47EL/action/replication_record"}},"created_at":"2026-07-05T05:59:51.750289+00:00","updated_at":"2026-07-05T05:59:51.750289+00:00"}