{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:3ZKS6EQ343LUACZ4JYGHYBVQ2Q","short_pith_number":"pith:3ZKS6EQ3","schema_version":"1.0","canonical_sha256":"de552f121be6d7400b3c4e0c7c06b0d438486644e004e86eb6cb51f3b5657b47","source":{"kind":"arxiv","id":"2205.09284","version":1},"attestation_state":"computed","paper":{"title":"Towards Applicable Reinforcement Learning: Improving the Generalization and Sample Efficiency with Policy Ensemble","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dongsheng Li, Jiang Bian, Kan Ren, Minghuan Liu, Weinan Zhang, Weiqing Liu, Xufang Luo, Zhengyu Yang","submitted_at":"2022-05-19T02:25:32Z","abstract_excerpt":"It is challenging for reinforcement learning (RL) algorithms to succeed in real-world applications like financial trading and logistic system due to the noisy observation and environment shifting between training and evaluation. Thus, it requires both high sample efficiency and generalization for resolving real-world tasks. However, directly applying typical RL algorithms can lead to poor performance in such scenarios. Considering the great performance of ensemble methods on both accuracy and generalization in supervised learning (SL), we design a robust and applicable method named Ensemble Pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.09284","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-05-19T02:25:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e5f6f6c4d44980c57ea119bc85bfb7f8dc903a22319ea7b899742610b2bfa29e","abstract_canon_sha256":"34a1ba06f050b190076d3eb5d76ca989251aa4b57742d65a8f1197ff428a25cd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:24:43.017803Z","signature_b64":"20Fs79tetfm0RZZ98Wuf8br1KiS7EhY8cuf+fBrec+YXYTITScXzqmkdAj3RumgkjFdX5OeHJVs2S5mHy7OvCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de552f121be6d7400b3c4e0c7c06b0d438486644e004e86eb6cb51f3b5657b47","last_reissued_at":"2026-07-05T04:24:43.017310Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:24:43.017310Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Applicable Reinforcement Learning: Improving the Generalization and Sample Efficiency with Policy Ensemble","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dongsheng Li, Jiang Bian, Kan Ren, Minghuan Liu, Weinan Zhang, Weiqing Liu, Xufang Luo, Zhengyu Yang","submitted_at":"2022-05-19T02:25:32Z","abstract_excerpt":"It is challenging for reinforcement learning (RL) algorithms to succeed in real-world applications like financial trading and logistic system due to the noisy observation and environment shifting between training and evaluation. Thus, it requires both high sample efficiency and generalization for resolving real-world tasks. However, directly applying typical RL algorithms can lead to poor performance in such scenarios. Considering the great performance of ensemble methods on both accuracy and generalization in supervised learning (SL), we design a robust and applicable method named Ensemble Pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.09284","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.09284/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.09284","created_at":"2026-07-05T04:24:43.017378+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.09284v1","created_at":"2026-07-05T04:24:43.017378+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.09284","created_at":"2026-07-05T04:24:43.017378+00:00"},{"alias_kind":"pith_short_12","alias_value":"3ZKS6EQ343LU","created_at":"2026-07-05T04:24:43.017378+00:00"},{"alias_kind":"pith_short_16","alias_value":"3ZKS6EQ343LUACZ4","created_at":"2026-07-05T04:24:43.017378+00:00"},{"alias_kind":"pith_short_8","alias_value":"3ZKS6EQ3","created_at":"2026-07-05T04:24:43.017378+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.03194","citing_title":"Scaling DRL for Decision Making: A Survey on Data, Network, and Training Budget Strategies","ref_index":56,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3ZKS6EQ343LUACZ4JYGHYBVQ2Q","json":"https://pith.science/pith/3ZKS6EQ343LUACZ4JYGHYBVQ2Q.json","graph_json":"https://pith.science/api/pith-number/3ZKS6EQ343LUACZ4JYGHYBVQ2Q/graph.json","events_json":"https://pith.science/api/pith-number/3ZKS6EQ343LUACZ4JYGHYBVQ2Q/events.json","paper":"https://pith.science/paper/3ZKS6EQ3"},"agent_actions":{"view_html":"https://pith.science/pith/3ZKS6EQ343LUACZ4JYGHYBVQ2Q","download_json":"https://pith.science/pith/3ZKS6EQ343LUACZ4JYGHYBVQ2Q.json","view_paper":"https://pith.science/paper/3ZKS6EQ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.09284&json=true","fetch_graph":"https://pith.science/api/pith-number/3ZKS6EQ343LUACZ4JYGHYBVQ2Q/graph.json","fetch_events":"https://pith.science/api/pith-number/3ZKS6EQ343LUACZ4JYGHYBVQ2Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3ZKS6EQ343LUACZ4JYGHYBVQ2Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3ZKS6EQ343LUACZ4JYGHYBVQ2Q/action/storage_attestation","attest_author":"https://pith.science/pith/3ZKS6EQ343LUACZ4JYGHYBVQ2Q/action/author_attestation","sign_citation":"https://pith.science/pith/3ZKS6EQ343LUACZ4JYGHYBVQ2Q/action/citation_signature","submit_replication":"https://pith.science/pith/3ZKS6EQ343LUACZ4JYGHYBVQ2Q/action/replication_record"}},"created_at":"2026-07-05T04:24:43.017378+00:00","updated_at":"2026-07-05T04:24:43.017378+00:00"}