{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:3UFOSJFEE6CGPDXHTSV4BQAZFI","short_pith_number":"pith:3UFOSJFE","schema_version":"1.0","canonical_sha256":"dd0ae924a42784678ee79cabc0c0192a0853a890b58871965d24fae1560e91dc","source":{"kind":"arxiv","id":"2204.12026","version":1},"attestation_state":"computed","paper":{"title":"BATS: Best Action Trajectory Stitching","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adam Villaflor, Ian Char, Jeff Schneider, John M. Dolan, Viraj Mehta","submitted_at":"2022-04-26T01:48:32Z","abstract_excerpt":"The problem of offline reinforcement learning focuses on learning a good policy from a log of environment interactions. Past efforts for developing algorithms in this area have revolved around introducing constraints to online reinforcement learning algorithms to ensure the actions of the learned policy are constrained to the logged data. In this work, we explore an alternative approach by planning on the fixed dataset directly. Specifically, we introduce an algorithm which forms a tabular Markov Decision Process (MDP) over the logged data by adding new transitions to the dataset. We do this b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.12026","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-04-26T01:48:32Z","cross_cats_sorted":[],"title_canon_sha256":"2f9dcc4a0f5ed5b0756a23812c5660b66680d39147cae15b63b8306d550c5e41","abstract_canon_sha256":"d7656fb520b3233962ac6bb479296d7ac2c95b50f987aefd35942129391c894a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:17:43.661713Z","signature_b64":"A4DxfCNA36BrrfKS/ALf8yY2IyjlChH23c0BIMUVziYFOy200RY8KSzN65xDjuUkbXxKKV7ScO1c4/6NoLltBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd0ae924a42784678ee79cabc0c0192a0853a890b58871965d24fae1560e91dc","last_reissued_at":"2026-07-05T04:17:43.661162Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:17:43.661162Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BATS: Best Action Trajectory Stitching","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adam Villaflor, Ian Char, Jeff Schneider, John M. Dolan, Viraj Mehta","submitted_at":"2022-04-26T01:48:32Z","abstract_excerpt":"The problem of offline reinforcement learning focuses on learning a good policy from a log of environment interactions. Past efforts for developing algorithms in this area have revolved around introducing constraints to online reinforcement learning algorithms to ensure the actions of the learned policy are constrained to the logged data. In this work, we explore an alternative approach by planning on the fixed dataset directly. Specifically, we introduce an algorithm which forms a tabular Markov Decision Process (MDP) over the logged data by adding new transitions to the dataset. We do this b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.12026","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.12026/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.12026","created_at":"2026-07-05T04:17:43.661222+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.12026v1","created_at":"2026-07-05T04:17:43.661222+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.12026","created_at":"2026-07-05T04:17:43.661222+00:00"},{"alias_kind":"pith_short_12","alias_value":"3UFOSJFEE6CG","created_at":"2026-07-05T04:17:43.661222+00:00"},{"alias_kind":"pith_short_16","alias_value":"3UFOSJFEE6CGPDXH","created_at":"2026-07-05T04:17:43.661222+00:00"},{"alias_kind":"pith_short_8","alias_value":"3UFOSJFE","created_at":"2026-07-05T04:17:43.661222+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21646","citing_title":"Energy-based Compositional Diffusion Planning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2506.10137","citing_title":"Self-Predictive Representations for Combinatorial Generalization in Behavioral Cloning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03075","citing_title":"Refining Compositional Diffusion for Reliable Long-Horizon Planning","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3UFOSJFEE6CGPDXHTSV4BQAZFI","json":"https://pith.science/pith/3UFOSJFEE6CGPDXHTSV4BQAZFI.json","graph_json":"https://pith.science/api/pith-number/3UFOSJFEE6CGPDXHTSV4BQAZFI/graph.json","events_json":"https://pith.science/api/pith-number/3UFOSJFEE6CGPDXHTSV4BQAZFI/events.json","paper":"https://pith.science/paper/3UFOSJFE"},"agent_actions":{"view_html":"https://pith.science/pith/3UFOSJFEE6CGPDXHTSV4BQAZFI","download_json":"https://pith.science/pith/3UFOSJFEE6CGPDXHTSV4BQAZFI.json","view_paper":"https://pith.science/paper/3UFOSJFE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.12026&json=true","fetch_graph":"https://pith.science/api/pith-number/3UFOSJFEE6CGPDXHTSV4BQAZFI/graph.json","fetch_events":"https://pith.science/api/pith-number/3UFOSJFEE6CGPDXHTSV4BQAZFI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3UFOSJFEE6CGPDXHTSV4BQAZFI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3UFOSJFEE6CGPDXHTSV4BQAZFI/action/storage_attestation","attest_author":"https://pith.science/pith/3UFOSJFEE6CGPDXHTSV4BQAZFI/action/author_attestation","sign_citation":"https://pith.science/pith/3UFOSJFEE6CGPDXHTSV4BQAZFI/action/citation_signature","submit_replication":"https://pith.science/pith/3UFOSJFEE6CGPDXHTSV4BQAZFI/action/replication_record"}},"created_at":"2026-07-05T04:17:43.661222+00:00","updated_at":"2026-07-05T04:17:43.661222+00:00"}