{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:I2AHBZDENJGPCHMDP3UOFFQBJJ","short_pith_number":"pith:I2AHBZDE","schema_version":"1.0","canonical_sha256":"468070e4646a4cf11d837ee8e296014a6ead4366a4246bd3fa008d026058944a","source":{"kind":"arxiv","id":"2411.07934","version":2},"attestation_state":"computed","paper":{"title":"Doubly Mild Generalization for Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Qi Wang, Xiangyang Ji, Yixiu Mao, Yuhang Jiang, Yun Qu","submitted_at":"2024-11-12T17:04:56Z","abstract_excerpt":"Offline Reinforcement Learning (RL) suffers from the extrapolation error and value overestimation. From a generalization perspective, this issue can be attributed to the over-generalization of value functions or policies towards out-of-distribution (OOD) actions. Significant efforts have been devoted to mitigating such generalization, and recent in-sample learning approaches have further succeeded in entirely eschewing it. Nevertheless, we show that mild generalization beyond the dataset can be trusted and leveraged to improve performance under certain conditions. To appropriately exploit gene"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.07934","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-11-12T17:04:56Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a93d42fabef5027ad2a5612190bc5ea963221720feb84d252ff68d9e8bf99290","abstract_canon_sha256":"80e0a65e3673234e0649c3f9b8529346f8b291452f677d206456ae6b734888ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:34:40.296114Z","signature_b64":"zQJhWphVH3gJ7YO5+d2w5CfjXAq9WG7lcvCH69FXuO8pHgQTVNS0AnMdCOroJpnc1ErIWCPCSBcMiwL1s1p5CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"468070e4646a4cf11d837ee8e296014a6ead4366a4246bd3fa008d026058944a","last_reissued_at":"2026-07-05T09:34:40.295630Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:34:40.295630Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Doubly Mild Generalization for Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Qi Wang, Xiangyang Ji, Yixiu Mao, Yuhang Jiang, Yun Qu","submitted_at":"2024-11-12T17:04:56Z","abstract_excerpt":"Offline Reinforcement Learning (RL) suffers from the extrapolation error and value overestimation. From a generalization perspective, this issue can be attributed to the over-generalization of value functions or policies towards out-of-distribution (OOD) actions. Significant efforts have been devoted to mitigating such generalization, and recent in-sample learning approaches have further succeeded in entirely eschewing it. Nevertheless, we show that mild generalization beyond the dataset can be trusted and leveraged to improve performance under certain conditions. To appropriately exploit gene"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.07934","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.07934/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.07934","created_at":"2026-07-05T09:34:40.295688+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.07934v2","created_at":"2026-07-05T09:34:40.295688+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.07934","created_at":"2026-07-05T09:34:40.295688+00:00"},{"alias_kind":"pith_short_12","alias_value":"I2AHBZDENJGP","created_at":"2026-07-05T09:34:40.295688+00:00"},{"alias_kind":"pith_short_16","alias_value":"I2AHBZDENJGPCHMD","created_at":"2026-07-05T09:34:40.295688+00:00"},{"alias_kind":"pith_short_8","alias_value":"I2AHBZDE","created_at":"2026-07-05T09:34:40.295688+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.19139","citing_title":"Fast and Robust: Task Sampling with Posterior and Diversity Synergies for Adaptive Decision-Makers in Randomized Environments","ref_index":50,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I2AHBZDENJGPCHMDP3UOFFQBJJ","json":"https://pith.science/pith/I2AHBZDENJGPCHMDP3UOFFQBJJ.json","graph_json":"https://pith.science/api/pith-number/I2AHBZDENJGPCHMDP3UOFFQBJJ/graph.json","events_json":"https://pith.science/api/pith-number/I2AHBZDENJGPCHMDP3UOFFQBJJ/events.json","paper":"https://pith.science/paper/I2AHBZDE"},"agent_actions":{"view_html":"https://pith.science/pith/I2AHBZDENJGPCHMDP3UOFFQBJJ","download_json":"https://pith.science/pith/I2AHBZDENJGPCHMDP3UOFFQBJJ.json","view_paper":"https://pith.science/paper/I2AHBZDE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.07934&json=true","fetch_graph":"https://pith.science/api/pith-number/I2AHBZDENJGPCHMDP3UOFFQBJJ/graph.json","fetch_events":"https://pith.science/api/pith-number/I2AHBZDENJGPCHMDP3UOFFQBJJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I2AHBZDENJGPCHMDP3UOFFQBJJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I2AHBZDENJGPCHMDP3UOFFQBJJ/action/storage_attestation","attest_author":"https://pith.science/pith/I2AHBZDENJGPCHMDP3UOFFQBJJ/action/author_attestation","sign_citation":"https://pith.science/pith/I2AHBZDENJGPCHMDP3UOFFQBJJ/action/citation_signature","submit_replication":"https://pith.science/pith/I2AHBZDENJGPCHMDP3UOFFQBJJ/action/replication_record"}},"created_at":"2026-07-05T09:34:40.295688+00:00","updated_at":"2026-07-05T09:34:40.295688+00:00"}