{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:VRHV5YRFQNA5WTHHI2GXKZZCMK","short_pith_number":"pith:VRHV5YRF","schema_version":"1.0","canonical_sha256":"ac4f5ee2258341db4ce7468d75672262b8fef44ebba7a2a4808a40834f0329e7","source":{"kind":"arxiv","id":"2206.03192","version":4},"attestation_state":"computed","paper":{"title":"Generalized Data Distribution Iteration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.NE"],"primary_cat":"cs.LG","authors_text":"Changnan Xiao, Jiajun Fan","submitted_at":"2022-06-07T11:27:40Z","abstract_excerpt":"To obtain higher sample efficiency and superior final performance simultaneously has been one of the major challenges for deep reinforcement learning (DRL). Previous work could handle one of these challenges but typically failed to address them concurrently. In this paper, we try to tackle these two challenges simultaneously. To achieve this, we firstly decouple these challenges into two classic RL problems: data richness and exploration-exploitation trade-off. Then, we cast these two problems into the training data distribution optimization problem, namely to obtain desired training data with"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.03192","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-07T11:27:40Z","cross_cats_sorted":["cs.AI","cs.NE"],"title_canon_sha256":"a4a57a67b249b39701fc75245f40934c50fd2f09bffc0e98045d4928224abba7","abstract_canon_sha256":"08b8bd3c72ac52ed6d516519a81de5d20600643955929178ebb865c1573f9bea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:33:00.557919Z","signature_b64":"AkWL/81zp66RTgAhldXEpz72VqJcH5276B/1oCJLw5u5zkwOq0A6FV6Qr+zQLl24xUuQXsW+Wv08Oxb+VxXlDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac4f5ee2258341db4ce7468d75672262b8fef44ebba7a2a4808a40834f0329e7","last_reissued_at":"2026-07-05T04:33:00.557495Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:33:00.557495Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalized Data Distribution Iteration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.NE"],"primary_cat":"cs.LG","authors_text":"Changnan Xiao, Jiajun Fan","submitted_at":"2022-06-07T11:27:40Z","abstract_excerpt":"To obtain higher sample efficiency and superior final performance simultaneously has been one of the major challenges for deep reinforcement learning (DRL). Previous work could handle one of these challenges but typically failed to address them concurrently. In this paper, we try to tackle these two challenges simultaneously. To achieve this, we firstly decouple these challenges into two classic RL problems: data richness and exploration-exploitation trade-off. Then, we cast these two problems into the training data distribution optimization problem, namely to obtain desired training data with"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.03192","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.03192/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.03192","created_at":"2026-07-05T04:33:00.557552+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.03192v4","created_at":"2026-07-05T04:33:00.557552+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.03192","created_at":"2026-07-05T04:33:00.557552+00:00"},{"alias_kind":"pith_short_12","alias_value":"VRHV5YRFQNA5","created_at":"2026-07-05T04:33:00.557552+00:00"},{"alias_kind":"pith_short_16","alias_value":"VRHV5YRFQNA5WTHH","created_at":"2026-07-05T04:33:00.557552+00:00"},{"alias_kind":"pith_short_8","alias_value":"VRHV5YRF","created_at":"2026-07-05T04:33:00.557552+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VRHV5YRFQNA5WTHHI2GXKZZCMK","json":"https://pith.science/pith/VRHV5YRFQNA5WTHHI2GXKZZCMK.json","graph_json":"https://pith.science/api/pith-number/VRHV5YRFQNA5WTHHI2GXKZZCMK/graph.json","events_json":"https://pith.science/api/pith-number/VRHV5YRFQNA5WTHHI2GXKZZCMK/events.json","paper":"https://pith.science/paper/VRHV5YRF"},"agent_actions":{"view_html":"https://pith.science/pith/VRHV5YRFQNA5WTHHI2GXKZZCMK","download_json":"https://pith.science/pith/VRHV5YRFQNA5WTHHI2GXKZZCMK.json","view_paper":"https://pith.science/paper/VRHV5YRF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.03192&json=true","fetch_graph":"https://pith.science/api/pith-number/VRHV5YRFQNA5WTHHI2GXKZZCMK/graph.json","fetch_events":"https://pith.science/api/pith-number/VRHV5YRFQNA5WTHHI2GXKZZCMK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VRHV5YRFQNA5WTHHI2GXKZZCMK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VRHV5YRFQNA5WTHHI2GXKZZCMK/action/storage_attestation","attest_author":"https://pith.science/pith/VRHV5YRFQNA5WTHHI2GXKZZCMK/action/author_attestation","sign_citation":"https://pith.science/pith/VRHV5YRFQNA5WTHHI2GXKZZCMK/action/citation_signature","submit_replication":"https://pith.science/pith/VRHV5YRFQNA5WTHHI2GXKZZCMK/action/replication_record"}},"created_at":"2026-07-05T04:33:00.557552+00:00","updated_at":"2026-07-05T04:33:00.557552+00:00"}