{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:VZON5FMNQXYZPTAYEXENUUHAFS","short_pith_number":"pith:VZON5FMN","schema_version":"1.0","canonical_sha256":"ae5cde958d85f197cc1825c8da50e02c8665e3512c5371302726c92b3dd614c1","source":{"kind":"arxiv","id":"2101.02649","version":1},"attestation_state":"computed","paper":{"title":"CoachNet: An Adversarial Sampling Approach for Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Elmira Amirloo Abolfathi, Jun Luo, Kasra Rezaee, Peyman Yadmellat","submitted_at":"2021-01-07T17:45:18Z","abstract_excerpt":"Despite the recent successes of reinforcement learning in games and robotics, it is yet to become broadly practical. Sample efficiency and unreliable performance in rare but challenging scenarios are two of the major obstacles. Drawing inspiration from the effectiveness of deliberate practice for achieving expert-level human performance, we propose a new adversarial sampling approach guided by a failure predictor named \"CoachNet\". CoachNet is trained online along with the agent to predict the probability of failure. This probability is then used in a stochastic sampling process to guide the ag"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.02649","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-01-07T17:45:18Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"798bef89e66698346b3e8ba98f0e3f2133ae795b03d160b313c6577b1513579c","abstract_canon_sha256":"ebb5770f63767958e21513a12b2530c73782d5e13ec5d67e6d8c70105ce3a966"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:05:15.742784Z","signature_b64":"QJibboBInk47aCyY8q0IxVB2ybPF+B/ihqzmJz/Kwz9wMgN8t9L5f/vsucHq4pZfxq7SD7MjCODZ/1CR0EBcAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae5cde958d85f197cc1825c8da50e02c8665e3512c5371302726c92b3dd614c1","last_reissued_at":"2026-07-05T02:05:15.742445Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:05:15.742445Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CoachNet: An Adversarial Sampling Approach for Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Elmira Amirloo Abolfathi, Jun Luo, Kasra Rezaee, Peyman Yadmellat","submitted_at":"2021-01-07T17:45:18Z","abstract_excerpt":"Despite the recent successes of reinforcement learning in games and robotics, it is yet to become broadly practical. Sample efficiency and unreliable performance in rare but challenging scenarios are two of the major obstacles. Drawing inspiration from the effectiveness of deliberate practice for achieving expert-level human performance, we propose a new adversarial sampling approach guided by a failure predictor named \"CoachNet\". CoachNet is trained online along with the agent to predict the probability of failure. This probability is then used in a stochastic sampling process to guide the ag"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.02649","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.02649/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.02649","created_at":"2026-07-05T02:05:15.742501+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.02649v1","created_at":"2026-07-05T02:05:15.742501+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.02649","created_at":"2026-07-05T02:05:15.742501+00:00"},{"alias_kind":"pith_short_12","alias_value":"VZON5FMNQXYZ","created_at":"2026-07-05T02:05:15.742501+00:00"},{"alias_kind":"pith_short_16","alias_value":"VZON5FMNQXYZPTAY","created_at":"2026-07-05T02:05:15.742501+00:00"},{"alias_kind":"pith_short_8","alias_value":"VZON5FMN","created_at":"2026-07-05T02:05:15.742501+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VZON5FMNQXYZPTAYEXENUUHAFS","json":"https://pith.science/pith/VZON5FMNQXYZPTAYEXENUUHAFS.json","graph_json":"https://pith.science/api/pith-number/VZON5FMNQXYZPTAYEXENUUHAFS/graph.json","events_json":"https://pith.science/api/pith-number/VZON5FMNQXYZPTAYEXENUUHAFS/events.json","paper":"https://pith.science/paper/VZON5FMN"},"agent_actions":{"view_html":"https://pith.science/pith/VZON5FMNQXYZPTAYEXENUUHAFS","download_json":"https://pith.science/pith/VZON5FMNQXYZPTAYEXENUUHAFS.json","view_paper":"https://pith.science/paper/VZON5FMN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.02649&json=true","fetch_graph":"https://pith.science/api/pith-number/VZON5FMNQXYZPTAYEXENUUHAFS/graph.json","fetch_events":"https://pith.science/api/pith-number/VZON5FMNQXYZPTAYEXENUUHAFS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VZON5FMNQXYZPTAYEXENUUHAFS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VZON5FMNQXYZPTAYEXENUUHAFS/action/storage_attestation","attest_author":"https://pith.science/pith/VZON5FMNQXYZPTAYEXENUUHAFS/action/author_attestation","sign_citation":"https://pith.science/pith/VZON5FMNQXYZPTAYEXENUUHAFS/action/citation_signature","submit_replication":"https://pith.science/pith/VZON5FMNQXYZPTAYEXENUUHAFS/action/replication_record"}},"created_at":"2026-07-05T02:05:15.742501+00:00","updated_at":"2026-07-05T02:05:15.742501+00:00"}