{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5OYUF6OLX47SZCCND63WDERS6E","short_pith_number":"pith:5OYUF6OL","schema_version":"1.0","canonical_sha256":"ebb142f9cbbf3f2c884d1fb7619232f131c96e81cdbc9390ee106eb060087515","source":{"kind":"arxiv","id":"2106.00808","version":4},"attestation_state":"computed","paper":{"title":"Invariant Policy Learning: A Causal Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jonas Peters, Niklas Pfister, Nikolaj Thams, Sorawit Saengkyongam","submitted_at":"2021-06-01T21:20:48Z","abstract_excerpt":"Contextual bandit and reinforcement learning algorithms have been successfully used in various interactive learning systems such as online advertising, recommender systems, and dynamic pricing. However, they have yet to be widely adopted in high-stakes application domains, such as healthcare. One reason may be that existing approaches assume that the underlying mechanisms are static in the sense that they do not change over different environments. In many real-world systems, however, the mechanisms are subject to shifts across environments which may invalidate the static environment assumption"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.00808","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-01T21:20:48Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"56c7def742d18977e9bb670f3075a79514389ffbd605778c0843b9a5131fc338","abstract_canon_sha256":"9748924f7756e4cab4824ce67314aa002adc908bb70bc1e651c73671c191c6d0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:59:57.240139Z","signature_b64":"p/bYs2lrwpw7lzB+VzUjTMlFG1JpmQ6Ro5VAnDH9d+utFlSwEE+Gggqsqf+piceepq48H3srTvGp8/+Lmmf1DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ebb142f9cbbf3f2c884d1fb7619232f131c96e81cdbc9390ee106eb060087515","last_reissued_at":"2026-07-05T04:59:57.239787Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:59:57.239787Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Invariant Policy Learning: A Causal Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jonas Peters, Niklas Pfister, Nikolaj Thams, Sorawit Saengkyongam","submitted_at":"2021-06-01T21:20:48Z","abstract_excerpt":"Contextual bandit and reinforcement learning algorithms have been successfully used in various interactive learning systems such as online advertising, recommender systems, and dynamic pricing. However, they have yet to be widely adopted in high-stakes application domains, such as healthcare. One reason may be that existing approaches assume that the underlying mechanisms are static in the sense that they do not change over different environments. In many real-world systems, however, the mechanisms are subject to shifts across environments which may invalidate the static environment assumption"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.00808","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.00808/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.00808","created_at":"2026-07-05T04:59:57.239842+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.00808v4","created_at":"2026-07-05T04:59:57.239842+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.00808","created_at":"2026-07-05T04:59:57.239842+00:00"},{"alias_kind":"pith_short_12","alias_value":"5OYUF6OLX47S","created_at":"2026-07-05T04:59:57.239842+00:00"},{"alias_kind":"pith_short_16","alias_value":"5OYUF6OLX47SZCCN","created_at":"2026-07-05T04:59:57.239842+00:00"},{"alias_kind":"pith_short_8","alias_value":"5OYUF6OL","created_at":"2026-07-05T04:59:57.239842+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5OYUF6OLX47SZCCND63WDERS6E","json":"https://pith.science/pith/5OYUF6OLX47SZCCND63WDERS6E.json","graph_json":"https://pith.science/api/pith-number/5OYUF6OLX47SZCCND63WDERS6E/graph.json","events_json":"https://pith.science/api/pith-number/5OYUF6OLX47SZCCND63WDERS6E/events.json","paper":"https://pith.science/paper/5OYUF6OL"},"agent_actions":{"view_html":"https://pith.science/pith/5OYUF6OLX47SZCCND63WDERS6E","download_json":"https://pith.science/pith/5OYUF6OLX47SZCCND63WDERS6E.json","view_paper":"https://pith.science/paper/5OYUF6OL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.00808&json=true","fetch_graph":"https://pith.science/api/pith-number/5OYUF6OLX47SZCCND63WDERS6E/graph.json","fetch_events":"https://pith.science/api/pith-number/5OYUF6OLX47SZCCND63WDERS6E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5OYUF6OLX47SZCCND63WDERS6E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5OYUF6OLX47SZCCND63WDERS6E/action/storage_attestation","attest_author":"https://pith.science/pith/5OYUF6OLX47SZCCND63WDERS6E/action/author_attestation","sign_citation":"https://pith.science/pith/5OYUF6OLX47SZCCND63WDERS6E/action/citation_signature","submit_replication":"https://pith.science/pith/5OYUF6OLX47SZCCND63WDERS6E/action/replication_record"}},"created_at":"2026-07-05T04:59:57.239842+00:00","updated_at":"2026-07-05T04:59:57.239842+00:00"}