{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:SYTSREGTAEEEABNGU66YUTYNKD","short_pith_number":"pith:SYTSREGT","schema_version":"1.0","canonical_sha256":"96272890d301084005a6a7bd8a4f0d50c6bda07d6196b475624456ff154eb698","source":{"kind":"arxiv","id":"2002.07418","version":2},"attestation_state":"computed","paper":{"title":"KoGuN: Accelerating Deep Reinforcement Learning via Integrating Human Suboptimal Knowledge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Hongyao Tang, Jianye Hao, Peng Zhang, Weixun Wang, Yan Zheng, Yihai Duan, Yi Ma","submitted_at":"2020-02-18T07:58:27Z","abstract_excerpt":"Reinforcement learning agents usually learn from scratch, which requires a large number of interactions with the environment. This is quite different from the learning process of human. When faced with a new task, human naturally have the common sense and use the prior knowledge to derive an initial policy and guide the learning process afterwards. Although the prior knowledge may be not fully applicable to the new task, the learning process is significantly sped up since the initial policy ensures a quick-start of learning and intermediate guidance allows to avoid unnecessary exploration. Tak"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.07418","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-02-18T07:58:27Z","cross_cats_sorted":[],"title_canon_sha256":"dba6f3ef2f00869dee53139486e722d9feb1fcd171c4ec72e5799dc7402e93c0","abstract_canon_sha256":"b07127cd52670da23b20d766a6f3a47165a6c774981dcce8a92d2929fbd1cfff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:04:48.174781Z","signature_b64":"0HAoA8CnF5QF50xSFQ9+IupErGIpnRr/9p5tVaRyZqXdrk8seRp3/samzPOc6gW88XnksVvoEB4RbIHKB/W2DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"96272890d301084005a6a7bd8a4f0d50c6bda07d6196b475624456ff154eb698","last_reissued_at":"2026-07-05T01:04:48.174036Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:04:48.174036Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KoGuN: Accelerating Deep Reinforcement Learning via Integrating Human Suboptimal Knowledge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Hongyao Tang, Jianye Hao, Peng Zhang, Weixun Wang, Yan Zheng, Yihai Duan, Yi Ma","submitted_at":"2020-02-18T07:58:27Z","abstract_excerpt":"Reinforcement learning agents usually learn from scratch, which requires a large number of interactions with the environment. This is quite different from the learning process of human. When faced with a new task, human naturally have the common sense and use the prior knowledge to derive an initial policy and guide the learning process afterwards. Although the prior knowledge may be not fully applicable to the new task, the learning process is significantly sped up since the initial policy ensures a quick-start of learning and intermediate guidance allows to avoid unnecessary exploration. Tak"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.07418","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.07418/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.07418","created_at":"2026-07-05T01:04:48.174099+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.07418v2","created_at":"2026-07-05T01:04:48.174099+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.07418","created_at":"2026-07-05T01:04:48.174099+00:00"},{"alias_kind":"pith_short_12","alias_value":"SYTSREGTAEEE","created_at":"2026-07-05T01:04:48.174099+00:00"},{"alias_kind":"pith_short_16","alias_value":"SYTSREGTAEEEABNG","created_at":"2026-07-05T01:04:48.174099+00:00"},{"alias_kind":"pith_short_8","alias_value":"SYTSREGT","created_at":"2026-07-05T01:04:48.174099+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12655","citing_title":"Robust Instruction Compliance in Cooperative Multi-Agent Reinforcement Learning","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SYTSREGTAEEEABNGU66YUTYNKD","json":"https://pith.science/pith/SYTSREGTAEEEABNGU66YUTYNKD.json","graph_json":"https://pith.science/api/pith-number/SYTSREGTAEEEABNGU66YUTYNKD/graph.json","events_json":"https://pith.science/api/pith-number/SYTSREGTAEEEABNGU66YUTYNKD/events.json","paper":"https://pith.science/paper/SYTSREGT"},"agent_actions":{"view_html":"https://pith.science/pith/SYTSREGTAEEEABNGU66YUTYNKD","download_json":"https://pith.science/pith/SYTSREGTAEEEABNGU66YUTYNKD.json","view_paper":"https://pith.science/paper/SYTSREGT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.07418&json=true","fetch_graph":"https://pith.science/api/pith-number/SYTSREGTAEEEABNGU66YUTYNKD/graph.json","fetch_events":"https://pith.science/api/pith-number/SYTSREGTAEEEABNGU66YUTYNKD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SYTSREGTAEEEABNGU66YUTYNKD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SYTSREGTAEEEABNGU66YUTYNKD/action/storage_attestation","attest_author":"https://pith.science/pith/SYTSREGTAEEEABNGU66YUTYNKD/action/author_attestation","sign_citation":"https://pith.science/pith/SYTSREGTAEEEABNGU66YUTYNKD/action/citation_signature","submit_replication":"https://pith.science/pith/SYTSREGTAEEEABNGU66YUTYNKD/action/replication_record"}},"created_at":"2026-07-05T01:04:48.174099+00:00","updated_at":"2026-07-05T01:04:48.174099+00:00"}