{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K7JDIDIELNXUD4VM77ZFBCA2ZW","short_pith_number":"pith:K7JDIDIE","schema_version":"1.0","canonical_sha256":"57d2340d045b6f41f2acfff250881acdb8ebd65be2647da57a23e5675325d462","source":{"kind":"arxiv","id":"2406.12905","version":1},"attestation_state":"computed","paper":{"title":"PufferLib: Making Reinforcement Learning Libraries and Environments Play Nice","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Joseph Suarez","submitted_at":"2024-06-11T21:13:34Z","abstract_excerpt":"You have an environment, a model, and a reinforcement learning library that are designed to work together but don't. PufferLib makes them play nice. The library provides one-line environment wrappers that eliminate common compatibility problems and fast vectorization to accelerate training. With PufferLib, you can use familiar libraries like CleanRL and SB3 to scale from classic benchmarks like Atari and Procgen to complex simulators like NetHack and Neural MMO. We release pip packages and prebuilt images with dependencies for dozens of environments. All of our code is free and open-source sof"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12905","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-11T21:13:34Z","cross_cats_sorted":["cs.AI","cs.MA"],"title_canon_sha256":"73ac831b103b6f95a44c4f2ec736bc2a31f5a955cf6b5e4265bb97264aa6c464","abstract_canon_sha256":"85e9942cf3aee003137e2f70ce58eb79dfde86ced5e6d5858d326c302d34dce4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:53.159078Z","signature_b64":"a3CTpwfCFlE/7EZVWz5ClNaYoXDbnsoIlY6LVJq67B66oSIhYnT7B6Fmr94TdyrF65iv/nkya8UdtuI/0reUAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57d2340d045b6f41f2acfff250881acdb8ebd65be2647da57a23e5675325d462","last_reissued_at":"2026-07-05T08:33:53.158572Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:53.158572Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PufferLib: Making Reinforcement Learning Libraries and Environments Play Nice","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Joseph Suarez","submitted_at":"2024-06-11T21:13:34Z","abstract_excerpt":"You have an environment, a model, and a reinforcement learning library that are designed to work together but don't. PufferLib makes them play nice. The library provides one-line environment wrappers that eliminate common compatibility problems and fast vectorization to accelerate training. With PufferLib, you can use familiar libraries like CleanRL and SB3 to scale from classic benchmarks like Atari and Procgen to complex simulators like NetHack and Neural MMO. We release pip packages and prebuilt images with dependencies for dozens of environments. All of our code is free and open-source sof"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12905","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12905/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12905","created_at":"2026-07-05T08:33:53.158646+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12905v1","created_at":"2026-07-05T08:33:53.158646+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12905","created_at":"2026-07-05T08:33:53.158646+00:00"},{"alias_kind":"pith_short_12","alias_value":"K7JDIDIELNXU","created_at":"2026-07-05T08:33:53.158646+00:00"},{"alias_kind":"pith_short_16","alias_value":"K7JDIDIELNXUD4VM","created_at":"2026-07-05T08:33:53.158646+00:00"},{"alias_kind":"pith_short_8","alias_value":"K7JDIDIE","created_at":"2026-07-05T08:33:53.158646+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19641","citing_title":"Scaling Self-Play for End-to-End Driving","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17386","citing_title":"TerraTransfer: Learning End-to-End Driving Policies Without Expert Demonstrations","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19370","citing_title":"Human-like autonomy emerges from self-play and a pinch of human data","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04149","citing_title":"CoPark: Learning Reactive Parking via Self-Play","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2603.12145","citing_title":"Automatic Generation of High-Performance RL Environments","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2509.00338","citing_title":"Scalable Option Learning in High-Throughput Environments","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27162","citing_title":"A High-Throughput Compute-Efficient POMDP Hide-And-Seek-Engine (HASE) for Multi-Agent Operations","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10910","citing_title":"Equivariant Reinforcement Learning for Clifford Quantum Circuit Synthesis","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2407.17032","citing_title":"Gymnasium: A Standard Interface for Reinforcement Learning Environments","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K7JDIDIELNXUD4VM77ZFBCA2ZW","json":"https://pith.science/pith/K7JDIDIELNXUD4VM77ZFBCA2ZW.json","graph_json":"https://pith.science/api/pith-number/K7JDIDIELNXUD4VM77ZFBCA2ZW/graph.json","events_json":"https://pith.science/api/pith-number/K7JDIDIELNXUD4VM77ZFBCA2ZW/events.json","paper":"https://pith.science/paper/K7JDIDIE"},"agent_actions":{"view_html":"https://pith.science/pith/K7JDIDIELNXUD4VM77ZFBCA2ZW","download_json":"https://pith.science/pith/K7JDIDIELNXUD4VM77ZFBCA2ZW.json","view_paper":"https://pith.science/paper/K7JDIDIE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12905&json=true","fetch_graph":"https://pith.science/api/pith-number/K7JDIDIELNXUD4VM77ZFBCA2ZW/graph.json","fetch_events":"https://pith.science/api/pith-number/K7JDIDIELNXUD4VM77ZFBCA2ZW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K7JDIDIELNXUD4VM77ZFBCA2ZW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K7JDIDIELNXUD4VM77ZFBCA2ZW/action/storage_attestation","attest_author":"https://pith.science/pith/K7JDIDIELNXUD4VM77ZFBCA2ZW/action/author_attestation","sign_citation":"https://pith.science/pith/K7JDIDIELNXUD4VM77ZFBCA2ZW/action/citation_signature","submit_replication":"https://pith.science/pith/K7JDIDIELNXUD4VM77ZFBCA2ZW/action/replication_record"}},"created_at":"2026-07-05T08:33:53.158646+00:00","updated_at":"2026-07-05T08:33:53.158646+00:00"}