{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:MWMXDSJRHJT2ZH3HNMWVDLZORZ","short_pith_number":"pith:MWMXDSJR","schema_version":"1.0","canonical_sha256":"659971c9313a67ac9f676b2d51af2e8e4597776e63f2757505dfc1ced841d695","source":{"kind":"arxiv","id":"2012.09092","version":1},"attestation_state":"computed","paper":{"title":"Sample-Efficient Reinforcement Learning via Counterfactual-Based Data Augmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Bernhard Sch\\\"olkopf, Biwei Huang, Chaochao Lu, Jos\\'e Miguel Hern\\'andez-Lobato, Ke Wang, Kun Zhang","submitted_at":"2020-12-16T17:21:13Z","abstract_excerpt":"Reinforcement learning (RL) algorithms usually require a substantial amount of interaction data and perform well only for specific tasks in a fixed environment. In some scenarios such as healthcare, however, usually only few records are available for each patient, and patients may show different responses to the same treatment, impeding the application of current RL algorithms to learn optimal policies. To address the issues of mechanism heterogeneity and related data scarcity, we propose a data-efficient RL algorithm that exploits structural causal models (SCMs) to model the state dynamics, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.09092","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-16T17:21:13Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"9589a01028cd87a59323040bf9f427d9f811fc8fba3057419ed38a1f6d0a53ef","abstract_canon_sha256":"ff86c4321afbd89324d8dee7a5748414566d375b3ff207d48d9f7d0ef1296bfe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:00:06.674253Z","signature_b64":"qeFotDl2o8haGBaeSTXfFwsy+gPmVzni6kEzyx2m1I4unicdrZ+0kUiljnxnEnxyEqH4x7fOEHUTvzH0q3ujAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"659971c9313a67ac9f676b2d51af2e8e4597776e63f2757505dfc1ced841d695","last_reissued_at":"2026-07-05T02:00:06.673815Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:00:06.673815Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sample-Efficient Reinforcement Learning via Counterfactual-Based Data Augmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Bernhard Sch\\\"olkopf, Biwei Huang, Chaochao Lu, Jos\\'e Miguel Hern\\'andez-Lobato, Ke Wang, Kun Zhang","submitted_at":"2020-12-16T17:21:13Z","abstract_excerpt":"Reinforcement learning (RL) algorithms usually require a substantial amount of interaction data and perform well only for specific tasks in a fixed environment. In some scenarios such as healthcare, however, usually only few records are available for each patient, and patients may show different responses to the same treatment, impeding the application of current RL algorithms to learn optimal policies. To address the issues of mechanism heterogeneity and related data scarcity, we propose a data-efficient RL algorithm that exploits structural causal models (SCMs) to model the state dynamics, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.09092","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.09092/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.09092","created_at":"2026-07-05T02:00:06.673877+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.09092v1","created_at":"2026-07-05T02:00:06.673877+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.09092","created_at":"2026-07-05T02:00:06.673877+00:00"},{"alias_kind":"pith_short_12","alias_value":"MWMXDSJRHJT2","created_at":"2026-07-05T02:00:06.673877+00:00"},{"alias_kind":"pith_short_16","alias_value":"MWMXDSJRHJT2ZH3H","created_at":"2026-07-05T02:00:06.673877+00:00"},{"alias_kind":"pith_short_8","alias_value":"MWMXDSJR","created_at":"2026-07-05T02:00:06.673877+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19460","citing_title":"Scaling Generative Foundation Models for Chest Radiography with Rectified Flow Transformers","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09115","citing_title":"Counterfactual Transport Flows for Offline Conservative Trajectory Refinement","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28460","citing_title":"Counterfactual Residual Data Augmentation for Regression","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13731","citing_title":"Robust Counterfactual Inference in Markov Decision Processes","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16054","citing_title":"Ada-Diffuser: Latent-Aware Adaptive Diffusion for Decision-Making","ref_index":234,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09030","citing_title":"PlayWorld: Learning Robot World Models from Autonomous Play","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MWMXDSJRHJT2ZH3HNMWVDLZORZ","json":"https://pith.science/pith/MWMXDSJRHJT2ZH3HNMWVDLZORZ.json","graph_json":"https://pith.science/api/pith-number/MWMXDSJRHJT2ZH3HNMWVDLZORZ/graph.json","events_json":"https://pith.science/api/pith-number/MWMXDSJRHJT2ZH3HNMWVDLZORZ/events.json","paper":"https://pith.science/paper/MWMXDSJR"},"agent_actions":{"view_html":"https://pith.science/pith/MWMXDSJRHJT2ZH3HNMWVDLZORZ","download_json":"https://pith.science/pith/MWMXDSJRHJT2ZH3HNMWVDLZORZ.json","view_paper":"https://pith.science/paper/MWMXDSJR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.09092&json=true","fetch_graph":"https://pith.science/api/pith-number/MWMXDSJRHJT2ZH3HNMWVDLZORZ/graph.json","fetch_events":"https://pith.science/api/pith-number/MWMXDSJRHJT2ZH3HNMWVDLZORZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MWMXDSJRHJT2ZH3HNMWVDLZORZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MWMXDSJRHJT2ZH3HNMWVDLZORZ/action/storage_attestation","attest_author":"https://pith.science/pith/MWMXDSJRHJT2ZH3HNMWVDLZORZ/action/author_attestation","sign_citation":"https://pith.science/pith/MWMXDSJRHJT2ZH3HNMWVDLZORZ/action/citation_signature","submit_replication":"https://pith.science/pith/MWMXDSJRHJT2ZH3HNMWVDLZORZ/action/replication_record"}},"created_at":"2026-07-05T02:00:06.673877+00:00","updated_at":"2026-07-05T02:00:06.673877+00:00"}