{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LH62F2VXKFHWTY3RR6TMKNBAPN","short_pith_number":"pith:LH62F2VX","schema_version":"1.0","canonical_sha256":"59fda2eab7514f69e3718fa6c534207b6b5c89f27ce53930f8593e0727ad4732","source":{"kind":"arxiv","id":"2406.02013","version":2},"attestation_state":"computed","paper":{"title":"Mamba as Decision Maker: Exploring Multi-scale Sequence Modeling in Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Gang Han, Hao Cheng, Jiahang Cao, Jiaxu Wang, Jingkai Sun, Qiang Zhang, Renjing Xu, Wen Zhao, Yecheng Shao, Yijie Guo, Ziqing Wang","submitted_at":"2024-06-04T06:49:18Z","abstract_excerpt":"Sequential modeling has demonstrated remarkable capabilities in offline reinforcement learning (RL), with Decision Transformer (DT) being one of the most notable representatives, achieving significant success. However, RL trajectories possess unique properties to be distinguished from the conventional sequence (e.g., text or audio): (1) local correlation, where the next states in RL are theoretically determined solely by current states and actions based on the Markov Decision Process (MDP), and (2) global correlation, where each step's features are related to long-term historical information d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.02013","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-04T06:49:18Z","cross_cats_sorted":[],"title_canon_sha256":"9a2b35a164a7f350756be6c1dcee3831909348184209ad776e227fafd240c97c","abstract_canon_sha256":"285c363b64c8fdcf1d92d17370ac685d1106bd1d9b513acc4f6e6be505f0aeb3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:05:36.412524Z","signature_b64":"ddsA1IKsLxt2jN4DfUusK8geMPFpSf5rpCHcYLKnCPuXOTHkFS27t0z0r9yT/BbaUnpqaWd79bPiTQ6o273sCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"59fda2eab7514f69e3718fa6c534207b6b5c89f27ce53930f8593e0727ad4732","last_reissued_at":"2026-07-05T09:05:36.412046Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:05:36.412046Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mamba as Decision Maker: Exploring Multi-scale Sequence Modeling in Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Gang Han, Hao Cheng, Jiahang Cao, Jiaxu Wang, Jingkai Sun, Qiang Zhang, Renjing Xu, Wen Zhao, Yecheng Shao, Yijie Guo, Ziqing Wang","submitted_at":"2024-06-04T06:49:18Z","abstract_excerpt":"Sequential modeling has demonstrated remarkable capabilities in offline reinforcement learning (RL), with Decision Transformer (DT) being one of the most notable representatives, achieving significant success. However, RL trajectories possess unique properties to be distinguished from the conventional sequence (e.g., text or audio): (1) local correlation, where the next states in RL are theoretically determined solely by current states and actions based on the Markov Decision Process (MDP), and (2) global correlation, where each step's features are related to long-term historical information d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.02013","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.02013/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.02013","created_at":"2026-07-05T09:05:36.412118+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.02013v2","created_at":"2026-07-05T09:05:36.412118+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.02013","created_at":"2026-07-05T09:05:36.412118+00:00"},{"alias_kind":"pith_short_12","alias_value":"LH62F2VXKFHW","created_at":"2026-07-05T09:05:36.412118+00:00"},{"alias_kind":"pith_short_16","alias_value":"LH62F2VXKFHWTY3R","created_at":"2026-07-05T09:05:36.412118+00:00"},{"alias_kind":"pith_short_8","alias_value":"LH62F2VX","created_at":"2026-07-05T09:05:36.412118+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2408.01129","citing_title":"A Survey of Mamba","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LH62F2VXKFHWTY3RR6TMKNBAPN","json":"https://pith.science/pith/LH62F2VXKFHWTY3RR6TMKNBAPN.json","graph_json":"https://pith.science/api/pith-number/LH62F2VXKFHWTY3RR6TMKNBAPN/graph.json","events_json":"https://pith.science/api/pith-number/LH62F2VXKFHWTY3RR6TMKNBAPN/events.json","paper":"https://pith.science/paper/LH62F2VX"},"agent_actions":{"view_html":"https://pith.science/pith/LH62F2VXKFHWTY3RR6TMKNBAPN","download_json":"https://pith.science/pith/LH62F2VXKFHWTY3RR6TMKNBAPN.json","view_paper":"https://pith.science/paper/LH62F2VX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.02013&json=true","fetch_graph":"https://pith.science/api/pith-number/LH62F2VXKFHWTY3RR6TMKNBAPN/graph.json","fetch_events":"https://pith.science/api/pith-number/LH62F2VXKFHWTY3RR6TMKNBAPN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LH62F2VXKFHWTY3RR6TMKNBAPN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LH62F2VXKFHWTY3RR6TMKNBAPN/action/storage_attestation","attest_author":"https://pith.science/pith/LH62F2VXKFHWTY3RR6TMKNBAPN/action/author_attestation","sign_citation":"https://pith.science/pith/LH62F2VXKFHWTY3RR6TMKNBAPN/action/citation_signature","submit_replication":"https://pith.science/pith/LH62F2VXKFHWTY3RR6TMKNBAPN/action/replication_record"}},"created_at":"2026-07-05T09:05:36.412118+00:00","updated_at":"2026-07-05T09:05:36.412118+00:00"}