{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:OG723QMIE3KTBBOU55UE2P2FQO","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"af3217682f85096b11f1cb473257f3d1d63c47de6706e2dc309bef3cdfa98451","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-14T16:30:03Z","title_canon_sha256":"00d3e16158c382a6644214edc66e45882030cce99db539618a382b7eefc696cb"},"schema_version":"1.0","source":{"id":"2405.08740","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.08740","created_at":"2026-07-05T08:26:24Z"},{"alias_kind":"arxiv_version","alias_value":"2405.08740v3","created_at":"2026-07-05T08:26:24Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.08740","created_at":"2026-07-05T08:26:24Z"},{"alias_kind":"pith_short_12","alias_value":"OG723QMIE3KT","created_at":"2026-07-05T08:26:24Z"},{"alias_kind":"pith_short_16","alias_value":"OG723QMIE3KTBBOU","created_at":"2026-07-05T08:26:24Z"},{"alias_kind":"pith_short_8","alias_value":"OG723QMI","created_at":"2026-07-05T08:26:24Z"}],"graph_snapshots":[{"event_id":"sha256:84ca718129dda3eb35169f3b764f3dd7facbfb3cbcd818e9a629726ce35f7ab7","target":"graph","created_at":"2026-07-05T08:26:24Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.08740/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"As a data-driven paradigm, offline reinforcement learning (RL) has been formulated as sequence modeling that conditions on the hindsight information including returns, goal or future trajectory. Although promising, this supervised paradigm overlooks the core objective of RL that maximizes the return. This overlook directly leads to the lack of trajectory stitching capability that affects the sequence model learning from sub-optimal data. In this work, we introduce the concept of max-return sequence modeling which integrates the goal of maximizing returns into existing sequence models. We propo","authors_text":"Dengyun Peng, Donglin Wang, Jinxin Liu, Zifeng Zhuang, Ziqi Zhang","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-14T16:30:03Z","title":"Reinformer: Max-Return Sequence Modeling for Offline RL"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.08740","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:75c152048f6c3dfe349563173cafb33d68c52d9cc2092f9777623b9daa79d63d","target":"record","created_at":"2026-07-05T08:26:24Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"af3217682f85096b11f1cb473257f3d1d63c47de6706e2dc309bef3cdfa98451","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-14T16:30:03Z","title_canon_sha256":"00d3e16158c382a6644214edc66e45882030cce99db539618a382b7eefc696cb"},"schema_version":"1.0","source":{"id":"2405.08740","kind":"arxiv","version":3}},"canonical_sha256":"71bfadc18826d53085d4ef684d3f4583b02e60810049d549898f219047dbf246","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"71bfadc18826d53085d4ef684d3f4583b02e60810049d549898f219047dbf246","first_computed_at":"2026-07-05T08:26:24.766057Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:26:24.766057Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Aq5P/Ae2YRHlfQDiWMYU2SrJhOlBWLnhXS4818Fkhn65T8GjxM8BmaZCo7Jq2ByoMpKH4z8HdQxxIEtQVt6GBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T08:26:24.766599Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.08740","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:75c152048f6c3dfe349563173cafb33d68c52d9cc2092f9777623b9daa79d63d","sha256:84ca718129dda3eb35169f3b764f3dd7facbfb3cbcd818e9a629726ce35f7ab7"],"state_sha256":"289dac4e7ad3cb64dafaa6daf06d8f76c3487845972423121d21a12901a670d6"}