{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IRUN6QO44RSNFK5R5Q3MEBHOFY","short_pith_number":"pith:IRUN6QO4","schema_version":"1.0","canonical_sha256":"4468df41dce464d2abb1ec36c204ee2e3ec983d9bbca96295b8f4db3dec0546b","source":{"kind":"arxiv","id":"2507.10174","version":1},"attestation_state":"computed","paper":{"title":"Should We Ever Prefer Decision Transformer for Offline Reinforcement Learning?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Keith Ross, Yumi Omori, Zixuan Dong","submitted_at":"2025-07-14T11:36:31Z","abstract_excerpt":"In recent years, extensive work has explored the application of the Transformer architecture to reinforcement learning problems. Among these, Decision Transformer (DT) has gained particular attention in the context of offline reinforcement learning due to its ability to frame return-conditioned policy learning as a sequence modeling task. Most recently, Bhargava et al. (2024) provided a systematic comparison of DT with more conventional MLP-based offline RL algorithms, including Behavior Cloning (BC) and Conservative Q-Learning (CQL), and claimed that DT exhibits superior performance in sparse"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.10174","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-14T11:36:31Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"84ba988f9ef53a6e915b33cdb14931e57f7d09a970880da27d7b0080007c697a","abstract_canon_sha256":"d20ae3627cf2fab481a6e480ec9789426856b21a2a7deca36fb77aba6f7fc5b0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:46.433020Z","signature_b64":"8yO51ijVHVho+zolv14E4EYrhaPKoWd3QwR+vJsLMfHbwjVvtDYytT9BkysKFx0E4aH0AkWoDEzk4UDtwbOLAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4468df41dce464d2abb1ec36c204ee2e3ec983d9bbca96295b8f4db3dec0546b","last_reissued_at":"2026-07-05T11:36:46.432572Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:46.432572Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Should We Ever Prefer Decision Transformer for Offline Reinforcement Learning?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Keith Ross, Yumi Omori, Zixuan Dong","submitted_at":"2025-07-14T11:36:31Z","abstract_excerpt":"In recent years, extensive work has explored the application of the Transformer architecture to reinforcement learning problems. Among these, Decision Transformer (DT) has gained particular attention in the context of offline reinforcement learning due to its ability to frame return-conditioned policy learning as a sequence modeling task. Most recently, Bhargava et al. (2024) provided a systematic comparison of DT with more conventional MLP-based offline RL algorithms, including Behavior Cloning (BC) and Conservative Q-Learning (CQL), and claimed that DT exhibits superior performance in sparse"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.10174","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.10174/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.10174","created_at":"2026-07-05T11:36:46.432633+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.10174v1","created_at":"2026-07-05T11:36:46.432633+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.10174","created_at":"2026-07-05T11:36:46.432633+00:00"},{"alias_kind":"pith_short_12","alias_value":"IRUN6QO44RSN","created_at":"2026-07-05T11:36:46.432633+00:00"},{"alias_kind":"pith_short_16","alias_value":"IRUN6QO44RSNFK5R","created_at":"2026-07-05T11:36:46.432633+00:00"},{"alias_kind":"pith_short_8","alias_value":"IRUN6QO4","created_at":"2026-07-05T11:36:46.432633+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IRUN6QO44RSNFK5R5Q3MEBHOFY","json":"https://pith.science/pith/IRUN6QO44RSNFK5R5Q3MEBHOFY.json","graph_json":"https://pith.science/api/pith-number/IRUN6QO44RSNFK5R5Q3MEBHOFY/graph.json","events_json":"https://pith.science/api/pith-number/IRUN6QO44RSNFK5R5Q3MEBHOFY/events.json","paper":"https://pith.science/paper/IRUN6QO4"},"agent_actions":{"view_html":"https://pith.science/pith/IRUN6QO44RSNFK5R5Q3MEBHOFY","download_json":"https://pith.science/pith/IRUN6QO44RSNFK5R5Q3MEBHOFY.json","view_paper":"https://pith.science/paper/IRUN6QO4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.10174&json=true","fetch_graph":"https://pith.science/api/pith-number/IRUN6QO44RSNFK5R5Q3MEBHOFY/graph.json","fetch_events":"https://pith.science/api/pith-number/IRUN6QO44RSNFK5R5Q3MEBHOFY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IRUN6QO44RSNFK5R5Q3MEBHOFY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IRUN6QO44RSNFK5R5Q3MEBHOFY/action/storage_attestation","attest_author":"https://pith.science/pith/IRUN6QO44RSNFK5R5Q3MEBHOFY/action/author_attestation","sign_citation":"https://pith.science/pith/IRUN6QO44RSNFK5R5Q3MEBHOFY/action/citation_signature","submit_replication":"https://pith.science/pith/IRUN6QO44RSNFK5R5Q3MEBHOFY/action/replication_record"}},"created_at":"2026-07-05T11:36:46.432633+00:00","updated_at":"2026-07-05T11:36:46.432633+00:00"}