{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XYPLQTMBANX6VUCRNVOZRHRXRE","short_pith_number":"pith:XYPLQTMB","schema_version":"1.0","canonical_sha256":"be1eb84d81036fead0516d5d989e37891af2e4410e47877299a8a021dcf36959","source":{"kind":"arxiv","id":"2111.03469","version":2},"attestation_state":"computed","paper":{"title":"Perturbational Complexity by Distribution Mismatch: A Systematic Analysis of Reinforcement Learning in Reproducing Kernel Hilbert Space","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jiequn Han, Jihao Long","submitted_at":"2021-11-05T12:46:04Z","abstract_excerpt":"Most existing theoretical analysis of reinforcement learning (RL) is limited to the tabular setting or linear models due to the difficulty in dealing with function approximation in high dimensional space with an uncertain environment. This work offers a fresh perspective into this challenge by analyzing RL in a general reproducing kernel Hilbert space (RKHS). We consider a family of Markov decision processes $\\mathcal{M}$ of which the reward functions lie in the unit ball of an RKHS and transition probabilities lie in a given arbitrary set. We define a quantity called perturbational complexity"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.03469","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-11-05T12:46:04Z","cross_cats_sorted":[],"title_canon_sha256":"94fbbd856845df9a48427ead3eaff11d2de86ee2304ef51c4d83cc6b87328a63","abstract_canon_sha256":"cf8fe014e66ebe55af07e0c1358491e5e1cb324993fbdd48a4c31e94677fe0ca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:08:59.456267Z","signature_b64":"HavvHkosbIR8uqNveU9frb07KFTKvjlOV/67nGBGMbhc7rsjlD1xDfwTGcttr9AoUXSFB9L+xjiU5fxUV3YPCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be1eb84d81036fead0516d5d989e37891af2e4410e47877299a8a021dcf36959","last_reissued_at":"2026-07-05T04:08:59.455808Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:08:59.455808Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Perturbational Complexity by Distribution Mismatch: A Systematic Analysis of Reinforcement Learning in Reproducing Kernel Hilbert Space","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jiequn Han, Jihao Long","submitted_at":"2021-11-05T12:46:04Z","abstract_excerpt":"Most existing theoretical analysis of reinforcement learning (RL) is limited to the tabular setting or linear models due to the difficulty in dealing with function approximation in high dimensional space with an uncertain environment. This work offers a fresh perspective into this challenge by analyzing RL in a general reproducing kernel Hilbert space (RKHS). We consider a family of Markov decision processes $\\mathcal{M}$ of which the reward functions lie in the unit ball of an RKHS and transition probabilities lie in a given arbitrary set. We define a quantity called perturbational complexity"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.03469","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.03469/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.03469","created_at":"2026-07-05T04:08:59.455867+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.03469v2","created_at":"2026-07-05T04:08:59.455867+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.03469","created_at":"2026-07-05T04:08:59.455867+00:00"},{"alias_kind":"pith_short_12","alias_value":"XYPLQTMBANX6","created_at":"2026-07-05T04:08:59.455867+00:00"},{"alias_kind":"pith_short_16","alias_value":"XYPLQTMBANX6VUCR","created_at":"2026-07-05T04:08:59.455867+00:00"},{"alias_kind":"pith_short_8","alias_value":"XYPLQTMB","created_at":"2026-07-05T04:08:59.455867+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XYPLQTMBANX6VUCRNVOZRHRXRE","json":"https://pith.science/pith/XYPLQTMBANX6VUCRNVOZRHRXRE.json","graph_json":"https://pith.science/api/pith-number/XYPLQTMBANX6VUCRNVOZRHRXRE/graph.json","events_json":"https://pith.science/api/pith-number/XYPLQTMBANX6VUCRNVOZRHRXRE/events.json","paper":"https://pith.science/paper/XYPLQTMB"},"agent_actions":{"view_html":"https://pith.science/pith/XYPLQTMBANX6VUCRNVOZRHRXRE","download_json":"https://pith.science/pith/XYPLQTMBANX6VUCRNVOZRHRXRE.json","view_paper":"https://pith.science/paper/XYPLQTMB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.03469&json=true","fetch_graph":"https://pith.science/api/pith-number/XYPLQTMBANX6VUCRNVOZRHRXRE/graph.json","fetch_events":"https://pith.science/api/pith-number/XYPLQTMBANX6VUCRNVOZRHRXRE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XYPLQTMBANX6VUCRNVOZRHRXRE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XYPLQTMBANX6VUCRNVOZRHRXRE/action/storage_attestation","attest_author":"https://pith.science/pith/XYPLQTMBANX6VUCRNVOZRHRXRE/action/author_attestation","sign_citation":"https://pith.science/pith/XYPLQTMBANX6VUCRNVOZRHRXRE/action/citation_signature","submit_replication":"https://pith.science/pith/XYPLQTMBANX6VUCRNVOZRHRXRE/action/replication_record"}},"created_at":"2026-07-05T04:08:59.455867+00:00","updated_at":"2026-07-05T04:08:59.455867+00:00"}