{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:5LVNYXPPE6P2NCHHSKTPIIJWMX","short_pith_number":"pith:5LVNYXPP","schema_version":"1.0","canonical_sha256":"eaeadc5def279fa688e792a6f4213665c580f97ac60f00fc5fb8b2e8215994d7","source":{"kind":"arxiv","id":"1811.06521","version":1},"attestation_state":"computed","paper":{"title":"Reward learning from human preferences and demonstrations in Atari","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.NE","stat.ML"],"primary_cat":"cs.LG","authors_text":"Borja Ibarz, Dario Amodei, Geoffrey Irving, Jan Leike, Shane Legg, Tobias Pohlen","submitted_at":"2018-11-15T18:33:43Z","abstract_excerpt":"To solve complex real-world problems with reinforcement learning, we cannot rely on manually specified reward functions. Instead, we can have humans communicate an objective to the agent directly. In this work, we combine two approaches to learning from human feedback: expert demonstrations and trajectory preferences. We train a deep neural network to model the reward function and use its predicted reward to train an DQN-based deep reinforcement learning agent on 9 Atari games. Our approach beats the imitation learning baseline in 7 games and achieves strictly superhuman performance on 2 games"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1811.06521","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2018-11-15T18:33:43Z","cross_cats_sorted":["cs.AI","cs.NE","stat.ML"],"title_canon_sha256":"d5cbc9f47ecd5078ddd8c2c5836a39d0bf4312273f02973d5944bafecbf99e64","abstract_canon_sha256":"4995f48a5890e07fc84e947aacfd04ca6b6dcf3ba30692fa6d77bb8de9621853"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:00:37.556192Z","signature_b64":"hp62vNurCvdFZuNE8E8UCR6yX1hqE2eMl35s3xKIwbu8WdhNKv/q785m56Yt9HuwRHjwfcFAaPjmFWMem89sCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eaeadc5def279fa688e792a6f4213665c580f97ac60f00fc5fb8b2e8215994d7","last_reissued_at":"2026-05-18T00:00:37.555718Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:00:37.555718Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reward learning from human preferences and demonstrations in Atari","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.NE","stat.ML"],"primary_cat":"cs.LG","authors_text":"Borja Ibarz, Dario Amodei, Geoffrey Irving, Jan Leike, Shane Legg, Tobias Pohlen","submitted_at":"2018-11-15T18:33:43Z","abstract_excerpt":"To solve complex real-world problems with reinforcement learning, we cannot rely on manually specified reward functions. Instead, we can have humans communicate an objective to the agent directly. In this work, we combine two approaches to learning from human feedback: expert demonstrations and trajectory preferences. We train a deep neural network to model the reward function and use its predicted reward to train an DQN-based deep reinforcement learning agent on 9 Atari games. Our approach beats the imitation learning baseline in 7 games and achieves strictly superhuman performance on 2 games"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1811.06521","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1811.06521","created_at":"2026-05-18T00:00:37.555787+00:00"},{"alias_kind":"arxiv_version","alias_value":"1811.06521v1","created_at":"2026-05-18T00:00:37.555787+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1811.06521","created_at":"2026-05-18T00:00:37.555787+00:00"},{"alias_kind":"pith_short_12","alias_value":"5LVNYXPPE6P2","created_at":"2026-05-18T12:32:08.215937+00:00"},{"alias_kind":"pith_short_16","alias_value":"5LVNYXPPE6P2NCHH","created_at":"2026-05-18T12:32:08.215937+00:00"},{"alias_kind":"pith_short_8","alias_value":"5LVNYXPP","created_at":"2026-05-18T12:32:08.215937+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"1906.01820","citing_title":"Risks from Learned Optimization in Advanced Machine Learning Systems","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2112.00861","citing_title":"A General Language Assistant as a Laboratory for Alignment","ref_index":229,"is_internal_anchor":false},{"citing_arxiv_id":"1909.08593","citing_title":"Fine-Tuning Language Models from Human Preferences","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5LVNYXPPE6P2NCHHSKTPIIJWMX","json":"https://pith.science/pith/5LVNYXPPE6P2NCHHSKTPIIJWMX.json","graph_json":"https://pith.science/api/pith-number/5LVNYXPPE6P2NCHHSKTPIIJWMX/graph.json","events_json":"https://pith.science/api/pith-number/5LVNYXPPE6P2NCHHSKTPIIJWMX/events.json","paper":"https://pith.science/paper/5LVNYXPP"},"agent_actions":{"view_html":"https://pith.science/pith/5LVNYXPPE6P2NCHHSKTPIIJWMX","download_json":"https://pith.science/pith/5LVNYXPPE6P2NCHHSKTPIIJWMX.json","view_paper":"https://pith.science/paper/5LVNYXPP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1811.06521&json=true","fetch_graph":"https://pith.science/api/pith-number/5LVNYXPPE6P2NCHHSKTPIIJWMX/graph.json","fetch_events":"https://pith.science/api/pith-number/5LVNYXPPE6P2NCHHSKTPIIJWMX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5LVNYXPPE6P2NCHHSKTPIIJWMX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5LVNYXPPE6P2NCHHSKTPIIJWMX/action/storage_attestation","attest_author":"https://pith.science/pith/5LVNYXPPE6P2NCHHSKTPIIJWMX/action/author_attestation","sign_citation":"https://pith.science/pith/5LVNYXPPE6P2NCHHSKTPIIJWMX/action/citation_signature","submit_replication":"https://pith.science/pith/5LVNYXPPE6P2NCHHSKTPIIJWMX/action/replication_record"}},"created_at":"2026-05-18T00:00:37.555787+00:00","updated_at":"2026-05-18T00:00:37.555787+00:00"}