{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:FQJ4SE5NLLS3VUF5SKKQROORF4","short_pith_number":"pith:FQJ4SE5N","schema_version":"1.0","canonical_sha256":"2c13c913ad5ae5bad0bd929508b9d12f13c0ad0dd3433876e6474e3b92dcab5b","source":{"kind":"arxiv","id":"1703.01703","version":2},"attestation_state":"computed","paper":{"title":"Third-Person Imitation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bradly C. Stadie, Ilya Sutskever, Pieter Abbeel","submitted_at":"2017-03-06T02:02:34Z","abstract_excerpt":"Reinforcement learning (RL) makes it possible to train agents capable of achieving sophisticated goals in complex and uncertain environments. A key difficulty in reinforcement learning is specifying a reward function for the agent to optimize. Traditionally, imitation learning in RL has been used to overcome this problem. Unfortunately, hitherto imitation learning methods tend to require that demonstrations are supplied in the first-person: the agent is provided with a sequence of states and a specification of the actions that it should have taken. While powerful, this kind of imitation learni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1703.01703","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2017-03-06T02:02:34Z","cross_cats_sorted":[],"title_canon_sha256":"58f1fc5e38153b76dacf46caea1d38ef3de9bf09a544a976ce80457b0d122d2a","abstract_canon_sha256":"3db0eaabd5ad4b141d2e0fed1d8fd3a81d22dcb69fde467d007ed918b91762e1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:06:03.312762Z","signature_b64":"CWM7t22aw9dy/pnEGd8ftdcdFLWdrL9rAi21V3dZA257d7IRzTTFQXsHNGBWCjOirZDLAB6cPih1wcmHl4SzDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c13c913ad5ae5bad0bd929508b9d12f13c0ad0dd3433876e6474e3b92dcab5b","last_reissued_at":"2026-07-05T00:06:03.312336Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:06:03.312336Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Third-Person Imitation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bradly C. Stadie, Ilya Sutskever, Pieter Abbeel","submitted_at":"2017-03-06T02:02:34Z","abstract_excerpt":"Reinforcement learning (RL) makes it possible to train agents capable of achieving sophisticated goals in complex and uncertain environments. A key difficulty in reinforcement learning is specifying a reward function for the agent to optimize. Traditionally, imitation learning in RL has been used to overcome this problem. Unfortunately, hitherto imitation learning methods tend to require that demonstrations are supplied in the first-person: the agent is provided with a sequence of states and a specification of the actions that it should have taken. While powerful, this kind of imitation learni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1703.01703","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1703.01703/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1703.01703","created_at":"2026-07-05T00:06:03.312412+00:00"},{"alias_kind":"arxiv_version","alias_value":"1703.01703v2","created_at":"2026-07-05T00:06:03.312412+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1703.01703","created_at":"2026-07-05T00:06:03.312412+00:00"},{"alias_kind":"pith_short_12","alias_value":"FQJ4SE5NLLS3","created_at":"2026-07-05T00:06:03.312412+00:00"},{"alias_kind":"pith_short_16","alias_value":"FQJ4SE5NLLS3VUF5","created_at":"2026-07-05T00:06:03.312412+00:00"},{"alias_kind":"pith_short_8","alias_value":"FQJ4SE5N","created_at":"2026-07-05T00:06:03.312412+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03201","citing_title":"Reinforcement Learning from Cross-domain Videos with Video Prediction Model","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"1906.10918","citing_title":"Towards Empathic Deep Q-Learning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2512.05335","citing_title":"State-Conditional Adversarial Learning: An Off-Policy Visual Domain Transfer Method for End-to-End Imitation Learning","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FQJ4SE5NLLS3VUF5SKKQROORF4","json":"https://pith.science/pith/FQJ4SE5NLLS3VUF5SKKQROORF4.json","graph_json":"https://pith.science/api/pith-number/FQJ4SE5NLLS3VUF5SKKQROORF4/graph.json","events_json":"https://pith.science/api/pith-number/FQJ4SE5NLLS3VUF5SKKQROORF4/events.json","paper":"https://pith.science/paper/FQJ4SE5N"},"agent_actions":{"view_html":"https://pith.science/pith/FQJ4SE5NLLS3VUF5SKKQROORF4","download_json":"https://pith.science/pith/FQJ4SE5NLLS3VUF5SKKQROORF4.json","view_paper":"https://pith.science/paper/FQJ4SE5N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1703.01703&json=true","fetch_graph":"https://pith.science/api/pith-number/FQJ4SE5NLLS3VUF5SKKQROORF4/graph.json","fetch_events":"https://pith.science/api/pith-number/FQJ4SE5NLLS3VUF5SKKQROORF4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FQJ4SE5NLLS3VUF5SKKQROORF4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FQJ4SE5NLLS3VUF5SKKQROORF4/action/storage_attestation","attest_author":"https://pith.science/pith/FQJ4SE5NLLS3VUF5SKKQROORF4/action/author_attestation","sign_citation":"https://pith.science/pith/FQJ4SE5NLLS3VUF5SKKQROORF4/action/citation_signature","submit_replication":"https://pith.science/pith/FQJ4SE5NLLS3VUF5SKKQROORF4/action/replication_record"}},"created_at":"2026-07-05T00:06:03.312412+00:00","updated_at":"2026-07-05T00:06:03.312412+00:00"}