{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:V5LEAEK6MLL33JRIK5UAP73T3Q","short_pith_number":"pith:V5LEAEK6","schema_version":"1.0","canonical_sha256":"af5640115e62d7bda628576807ff73dc03c055f733e322242db3017860baa2e6","source":{"kind":"arxiv","id":"2208.11040","version":1},"attestation_state":"computed","paper":{"title":"Strategic Decision-Making in the Presence of Information Asymmetry: Provably Efficient RL with Algorithmic Instruments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IT","cs.LG","math.IT","math.OC","stat.ME"],"primary_cat":"stat.ML","authors_text":"Jianqing Fan, Mengxin Yu, Zhuoran Yang","submitted_at":"2022-08-23T15:32:44Z","abstract_excerpt":"We study offline reinforcement learning under a novel model called strategic MDP, which characterizes the strategic interactions between a principal and a sequence of myopic agents with private types. Due to the bilevel structure and private types, strategic MDP involves information asymmetry between the principal and the agents. We focus on the offline RL problem, where the goal is to learn the optimal policy of the principal concerning a target population of agents based on a pre-collected dataset that consists of historical interactions. The unobserved private types confound such a dataset "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.11040","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2022-08-23T15:32:44Z","cross_cats_sorted":["cs.IT","cs.LG","math.IT","math.OC","stat.ME"],"title_canon_sha256":"b549ed77ed8e3e8cd2a6cb3a25dff8f779787519f1497144b568ad6b1ec7ec49","abstract_canon_sha256":"6f9e49f5a35aff5ef81d9707e4d244a79b330376ace412be889706c72550de3d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:50:43.215415Z","signature_b64":"PxlfX4zgev4GNaqZqKqW5iFXEuDg9+jZns2A8ZsIw0Bv6XS0w8PM6NBr/FeY/YLIR3dsb7a1xf3rX4i4E3JVDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af5640115e62d7bda628576807ff73dc03c055f733e322242db3017860baa2e6","last_reissued_at":"2026-07-05T04:50:43.214922Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:50:43.214922Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Strategic Decision-Making in the Presence of Information Asymmetry: Provably Efficient RL with Algorithmic Instruments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IT","cs.LG","math.IT","math.OC","stat.ME"],"primary_cat":"stat.ML","authors_text":"Jianqing Fan, Mengxin Yu, Zhuoran Yang","submitted_at":"2022-08-23T15:32:44Z","abstract_excerpt":"We study offline reinforcement learning under a novel model called strategic MDP, which characterizes the strategic interactions between a principal and a sequence of myopic agents with private types. Due to the bilevel structure and private types, strategic MDP involves information asymmetry between the principal and the agents. We focus on the offline RL problem, where the goal is to learn the optimal policy of the principal concerning a target population of agents based on a pre-collected dataset that consists of historical interactions. The unobserved private types confound such a dataset "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.11040","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.11040/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.11040","created_at":"2026-07-05T04:50:43.214980+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.11040v1","created_at":"2026-07-05T04:50:43.214980+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.11040","created_at":"2026-07-05T04:50:43.214980+00:00"},{"alias_kind":"pith_short_12","alias_value":"V5LEAEK6MLL3","created_at":"2026-07-05T04:50:43.214980+00:00"},{"alias_kind":"pith_short_16","alias_value":"V5LEAEK6MLL33JRI","created_at":"2026-07-05T04:50:43.214980+00:00"},{"alias_kind":"pith_short_8","alias_value":"V5LEAEK6","created_at":"2026-07-05T04:50:43.214980+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2405.20642","citing_title":"Learning Under Moral Hazard with Instrumental Regression and Generalized Method of Moments","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V5LEAEK6MLL33JRIK5UAP73T3Q","json":"https://pith.science/pith/V5LEAEK6MLL33JRIK5UAP73T3Q.json","graph_json":"https://pith.science/api/pith-number/V5LEAEK6MLL33JRIK5UAP73T3Q/graph.json","events_json":"https://pith.science/api/pith-number/V5LEAEK6MLL33JRIK5UAP73T3Q/events.json","paper":"https://pith.science/paper/V5LEAEK6"},"agent_actions":{"view_html":"https://pith.science/pith/V5LEAEK6MLL33JRIK5UAP73T3Q","download_json":"https://pith.science/pith/V5LEAEK6MLL33JRIK5UAP73T3Q.json","view_paper":"https://pith.science/paper/V5LEAEK6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.11040&json=true","fetch_graph":"https://pith.science/api/pith-number/V5LEAEK6MLL33JRIK5UAP73T3Q/graph.json","fetch_events":"https://pith.science/api/pith-number/V5LEAEK6MLL33JRIK5UAP73T3Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V5LEAEK6MLL33JRIK5UAP73T3Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V5LEAEK6MLL33JRIK5UAP73T3Q/action/storage_attestation","attest_author":"https://pith.science/pith/V5LEAEK6MLL33JRIK5UAP73T3Q/action/author_attestation","sign_citation":"https://pith.science/pith/V5LEAEK6MLL33JRIK5UAP73T3Q/action/citation_signature","submit_replication":"https://pith.science/pith/V5LEAEK6MLL33JRIK5UAP73T3Q/action/replication_record"}},"created_at":"2026-07-05T04:50:43.214980+00:00","updated_at":"2026-07-05T04:50:43.214980+00:00"}