{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:MWMXDSJRHJT2ZH3HNMWVDLZORZ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ff86c4321afbd89324d8dee7a5748414566d375b3ff207d48d9f7d0ef1296bfe","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-16T17:21:13Z","title_canon_sha256":"9589a01028cd87a59323040bf9f427d9f811fc8fba3057419ed38a1f6d0a53ef"},"schema_version":"1.0","source":{"id":"2012.09092","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2012.09092","created_at":"2026-07-05T02:00:06Z"},{"alias_kind":"arxiv_version","alias_value":"2012.09092v1","created_at":"2026-07-05T02:00:06Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.09092","created_at":"2026-07-05T02:00:06Z"},{"alias_kind":"pith_short_12","alias_value":"MWMXDSJRHJT2","created_at":"2026-07-05T02:00:06Z"},{"alias_kind":"pith_short_16","alias_value":"MWMXDSJRHJT2ZH3H","created_at":"2026-07-05T02:00:06Z"},{"alias_kind":"pith_short_8","alias_value":"MWMXDSJR","created_at":"2026-07-05T02:00:06Z"}],"graph_snapshots":[{"event_id":"sha256:45df4af9845950638b32143add163b12c96cf6f58d0aa85bae5bb7d03a308692","target":"graph","created_at":"2026-07-05T02:00:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2012.09092/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) algorithms usually require a substantial amount of interaction data and perform well only for specific tasks in a fixed environment. In some scenarios such as healthcare, however, usually only few records are available for each patient, and patients may show different responses to the same treatment, impeding the application of current RL algorithms to learn optimal policies. To address the issues of mechanism heterogeneity and related data scarcity, we propose a data-efficient RL algorithm that exploits structural causal models (SCMs) to model the state dynamics, w","authors_text":"Bernhard Sch\\\"olkopf, Biwei Huang, Chaochao Lu, Jos\\'e Miguel Hern\\'andez-Lobato, Ke Wang, Kun Zhang","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-16T17:21:13Z","title":"Sample-Efficient Reinforcement Learning via Counterfactual-Based Data Augmentation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.09092","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:8747aa3951ac09d385944a99a8b232eb18ae239c9759c484bd88a02e51448b31","target":"record","created_at":"2026-07-05T02:00:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ff86c4321afbd89324d8dee7a5748414566d375b3ff207d48d9f7d0ef1296bfe","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-16T17:21:13Z","title_canon_sha256":"9589a01028cd87a59323040bf9f427d9f811fc8fba3057419ed38a1f6d0a53ef"},"schema_version":"1.0","source":{"id":"2012.09092","kind":"arxiv","version":1}},"canonical_sha256":"659971c9313a67ac9f676b2d51af2e8e4597776e63f2757505dfc1ced841d695","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"659971c9313a67ac9f676b2d51af2e8e4597776e63f2757505dfc1ced841d695","first_computed_at":"2026-07-05T02:00:06.673815Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:00:06.673815Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"qeFotDl2o8haGBaeSTXfFwsy+gPmVzni6kEzyx2m1I4unicdrZ+0kUiljnxnEnxyEqH4x7fOEHUTvzH0q3ujAw==","signature_status":"signed_v1","signed_at":"2026-07-05T02:00:06.674253Z","signed_message":"canonical_sha256_bytes"},"source_id":"2012.09092","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:8747aa3951ac09d385944a99a8b232eb18ae239c9759c484bd88a02e51448b31","sha256:45df4af9845950638b32143add163b12c96cf6f58d0aa85bae5bb7d03a308692"],"state_sha256":"de50ea73b2282b0ccb12f34968e0cec9c28b1dc1d4cb706759bf1c38143c73e8"}