{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:H7VNZEODLCU245HUSFC2PK3CNA","short_pith_number":"pith:H7VNZEOD","schema_version":"1.0","canonical_sha256":"3feadc91c358a9ae74f49145a7ab62683068bd07337a9612e51cc5d24373b866","source":{"kind":"arxiv","id":"2505.19058","version":1},"attestation_state":"computed","paper":{"title":"Distributionally Robust Deep Q-Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","q-fin.PM","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aijia Zhang, Chung I Lu, Julian Sester","submitted_at":"2025-05-25T09:22:06Z","abstract_excerpt":"We propose a novel distributionally robust $Q$-learning algorithm for the non-tabular case accounting for continuous state spaces where the state transition of the underlying Markov decision process is subject to model uncertainty. The uncertainty is taken into account by considering the worst-case transition from a ball around a reference probability measure. To determine the optimal policy under the worst-case state transition, we solve the associated non-linear Bellman equation by dualising and regularising the Bellman operator with the Sinkhorn distance, which is then parameterized with de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.19058","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-25T09:22:06Z","cross_cats_sorted":["math.OC","q-fin.PM","stat.ML"],"title_canon_sha256":"05a90620a18ef27cac2a12fefe607424a5e79fcc029404012ab975f1026a7c6f","abstract_canon_sha256":"28a13b6557b1b1d4b3be46c581f29a1fee84131fd1785840300e01bd4fa505ca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:24.240615Z","signature_b64":"k7MT+XkVPYBQCTpsAtHwlX/O1btPPQyR3uI0FWSMlrrA2DDe/aaIKk/C+yN2d/yXiWBScPQ9IpUHrO72T5ldDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3feadc91c358a9ae74f49145a7ab62683068bd07337a9612e51cc5d24373b866","last_reissued_at":"2026-07-05T11:09:24.240116Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:24.240116Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distributionally Robust Deep Q-Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","q-fin.PM","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aijia Zhang, Chung I Lu, Julian Sester","submitted_at":"2025-05-25T09:22:06Z","abstract_excerpt":"We propose a novel distributionally robust $Q$-learning algorithm for the non-tabular case accounting for continuous state spaces where the state transition of the underlying Markov decision process is subject to model uncertainty. The uncertainty is taken into account by considering the worst-case transition from a ball around a reference probability measure. To determine the optimal policy under the worst-case state transition, we solve the associated non-linear Bellman equation by dualising and regularising the Bellman operator with the Sinkhorn distance, which is then parameterized with de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.19058","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.19058/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.19058","created_at":"2026-07-05T11:09:24.240173+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.19058v1","created_at":"2026-07-05T11:09:24.240173+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.19058","created_at":"2026-07-05T11:09:24.240173+00:00"},{"alias_kind":"pith_short_12","alias_value":"H7VNZEODLCU2","created_at":"2026-07-05T11:09:24.240173+00:00"},{"alias_kind":"pith_short_16","alias_value":"H7VNZEODLCU245HU","created_at":"2026-07-05T11:09:24.240173+00:00"},{"alias_kind":"pith_short_8","alias_value":"H7VNZEOD","created_at":"2026-07-05T11:09:24.240173+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08291","citing_title":"Robustness in Sequential Decision Making under Evolving Uncertainty: Evidence from High-Frequency Market Making","ref_index":105,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20356","citing_title":"Robust $Q$-learning for mean-field control under Wasserstein uncertainty in common noise","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H7VNZEODLCU245HUSFC2PK3CNA","json":"https://pith.science/pith/H7VNZEODLCU245HUSFC2PK3CNA.json","graph_json":"https://pith.science/api/pith-number/H7VNZEODLCU245HUSFC2PK3CNA/graph.json","events_json":"https://pith.science/api/pith-number/H7VNZEODLCU245HUSFC2PK3CNA/events.json","paper":"https://pith.science/paper/H7VNZEOD"},"agent_actions":{"view_html":"https://pith.science/pith/H7VNZEODLCU245HUSFC2PK3CNA","download_json":"https://pith.science/pith/H7VNZEODLCU245HUSFC2PK3CNA.json","view_paper":"https://pith.science/paper/H7VNZEOD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.19058&json=true","fetch_graph":"https://pith.science/api/pith-number/H7VNZEODLCU245HUSFC2PK3CNA/graph.json","fetch_events":"https://pith.science/api/pith-number/H7VNZEODLCU245HUSFC2PK3CNA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H7VNZEODLCU245HUSFC2PK3CNA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H7VNZEODLCU245HUSFC2PK3CNA/action/storage_attestation","attest_author":"https://pith.science/pith/H7VNZEODLCU245HUSFC2PK3CNA/action/author_attestation","sign_citation":"https://pith.science/pith/H7VNZEODLCU245HUSFC2PK3CNA/action/citation_signature","submit_replication":"https://pith.science/pith/H7VNZEODLCU245HUSFC2PK3CNA/action/replication_record"}},"created_at":"2026-07-05T11:09:24.240173+00:00","updated_at":"2026-07-05T11:09:24.240173+00:00"}