{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:XAOBWPMSTQBUQ74NS3DAERIKMQ","short_pith_number":"pith:XAOBWPMS","schema_version":"1.0","canonical_sha256":"b81c1b3d929c03487f8d96c602450a642a64bd10714f3752472819449d6f2da6","source":{"kind":"arxiv","id":"2011.11827","version":3},"attestation_state":"computed","paper":{"title":"REPAINT: Knowledge Transfer in Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Jonathan Chung, Sahika Genc, Sunil Mallya, Tao Sun, Yunzhe Tao","submitted_at":"2020-11-24T01:18:32Z","abstract_excerpt":"Accelerating learning processes for complex tasks by leveraging previously learned tasks has been one of the most challenging problems in reinforcement learning, especially when the similarity between source and target tasks is low. This work proposes REPresentation And INstance Transfer (REPAINT) algorithm for knowledge transfer in deep reinforcement learning. REPAINT not only transfers the representation of a pre-trained teacher policy in the on-policy learning, but also uses an advantage-based experience selection approach to transfer useful samples collected following the teacher policy in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.11827","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-11-24T01:18:32Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"c62dc1b6d0701f0ecf58f59b47e60e3cc3baeb5c43c719e5c2f3e788b168c094","abstract_canon_sha256":"b7aac46619300405c969323c2bda59ba6495ecc769a07936f8da8506fd012a3b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:43:30.948459Z","signature_b64":"nFsmkfZkQiT63FqEQOzJjwzvElySxjYxzKr4jxHftFA5P2ywdlCbhye4jEuJrDyHEBTwVHFOLC5Lp35EE/BjDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b81c1b3d929c03487f8d96c602450a642a64bd10714f3752472819449d6f2da6","last_reissued_at":"2026-07-05T02:43:30.948057Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:43:30.948057Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"REPAINT: Knowledge Transfer in Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Jonathan Chung, Sahika Genc, Sunil Mallya, Tao Sun, Yunzhe Tao","submitted_at":"2020-11-24T01:18:32Z","abstract_excerpt":"Accelerating learning processes for complex tasks by leveraging previously learned tasks has been one of the most challenging problems in reinforcement learning, especially when the similarity between source and target tasks is low. This work proposes REPresentation And INstance Transfer (REPAINT) algorithm for knowledge transfer in deep reinforcement learning. REPAINT not only transfers the representation of a pre-trained teacher policy in the on-policy learning, but also uses an advantage-based experience selection approach to transfer useful samples collected following the teacher policy in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.11827","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.11827/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.11827","created_at":"2026-07-05T02:43:30.948126+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.11827v3","created_at":"2026-07-05T02:43:30.948126+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.11827","created_at":"2026-07-05T02:43:30.948126+00:00"},{"alias_kind":"pith_short_12","alias_value":"XAOBWPMSTQBU","created_at":"2026-07-05T02:43:30.948126+00:00"},{"alias_kind":"pith_short_16","alias_value":"XAOBWPMSTQBUQ74N","created_at":"2026-07-05T02:43:30.948126+00:00"},{"alias_kind":"pith_short_8","alias_value":"XAOBWPMS","created_at":"2026-07-05T02:43:30.948126+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XAOBWPMSTQBUQ74NS3DAERIKMQ","json":"https://pith.science/pith/XAOBWPMSTQBUQ74NS3DAERIKMQ.json","graph_json":"https://pith.science/api/pith-number/XAOBWPMSTQBUQ74NS3DAERIKMQ/graph.json","events_json":"https://pith.science/api/pith-number/XAOBWPMSTQBUQ74NS3DAERIKMQ/events.json","paper":"https://pith.science/paper/XAOBWPMS"},"agent_actions":{"view_html":"https://pith.science/pith/XAOBWPMSTQBUQ74NS3DAERIKMQ","download_json":"https://pith.science/pith/XAOBWPMSTQBUQ74NS3DAERIKMQ.json","view_paper":"https://pith.science/paper/XAOBWPMS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.11827&json=true","fetch_graph":"https://pith.science/api/pith-number/XAOBWPMSTQBUQ74NS3DAERIKMQ/graph.json","fetch_events":"https://pith.science/api/pith-number/XAOBWPMSTQBUQ74NS3DAERIKMQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XAOBWPMSTQBUQ74NS3DAERIKMQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XAOBWPMSTQBUQ74NS3DAERIKMQ/action/storage_attestation","attest_author":"https://pith.science/pith/XAOBWPMSTQBUQ74NS3DAERIKMQ/action/author_attestation","sign_citation":"https://pith.science/pith/XAOBWPMSTQBUQ74NS3DAERIKMQ/action/citation_signature","submit_replication":"https://pith.science/pith/XAOBWPMSTQBUQ74NS3DAERIKMQ/action/replication_record"}},"created_at":"2026-07-05T02:43:30.948126+00:00","updated_at":"2026-07-05T02:43:30.948126+00:00"}