{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:GEZQSGYMFQO6RWRD7XNUWZT6I5","short_pith_number":"pith:GEZQSGYM","canonical_record":{"source":{"id":"2408.15593","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-28T07:36:20Z","cross_cats_sorted":[],"title_canon_sha256":"dfa16a5680c6f12258629318ad146d23d2c527f0734dfba939494fc8a21e439d","abstract_canon_sha256":"54099a741ec3613027cdb8f26f68614408ce7fbd909105431cd25d25a112ff04"},"schema_version":"1.0"},"canonical_sha256":"3133091b0c2c1de8da23fddb4b667e4776c6d97edfdda6795266e6610ed4c97c","source":{"kind":"arxiv","id":"2408.15593","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2408.15593","created_at":"2026-07-05T09:00:14Z"},{"alias_kind":"arxiv_version","alias_value":"2408.15593v1","created_at":"2026-07-05T09:00:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.15593","created_at":"2026-07-05T09:00:14Z"},{"alias_kind":"pith_short_12","alias_value":"GEZQSGYMFQO6","created_at":"2026-07-05T09:00:14Z"},{"alias_kind":"pith_short_16","alias_value":"GEZQSGYMFQO6RWRD","created_at":"2026-07-05T09:00:14Z"},{"alias_kind":"pith_short_8","alias_value":"GEZQSGYM","created_at":"2026-07-05T09:00:14Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:GEZQSGYMFQO6RWRD7XNUWZT6I5","target":"record","payload":{"canonical_record":{"source":{"id":"2408.15593","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-28T07:36:20Z","cross_cats_sorted":[],"title_canon_sha256":"dfa16a5680c6f12258629318ad146d23d2c527f0734dfba939494fc8a21e439d","abstract_canon_sha256":"54099a741ec3613027cdb8f26f68614408ce7fbd909105431cd25d25a112ff04"},"schema_version":"1.0"},"canonical_sha256":"3133091b0c2c1de8da23fddb4b667e4776c6d97edfdda6795266e6610ed4c97c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:00:14.258426Z","signature_b64":"fF1b1X6ddcWKCLXeKk7jpkXanKLINSz8hFyDsbx0lllu0LdWykADmNmFkqT+4VldaBcThtO7hHOs2oFErXhDBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3133091b0c2c1de8da23fddb4b667e4776c6d97edfdda6795266e6610ed4c97c","last_reissued_at":"2026-07-05T09:00:14.257934Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:00:14.257934Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2408.15593","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:00:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JMPQ7EX7l7e0siiw44ownXLnvCqpSIcxB++pscRjp/5pync2iI8Qwcf8Ycxh6Dd6MR5Urs99xh0nHJL3r2j4Aw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T06:27:18.726608Z"},"content_sha256":"25708fac50fbf67f1000a649cc49138c8ebbfa2542289ea01adadb01063199a3","schema_version":"1.0","event_id":"sha256:25708fac50fbf67f1000a649cc49138c8ebbfa2542289ea01adadb01063199a3"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:GEZQSGYMFQO6RWRD7XNUWZT6I5","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Skills Regularized Task Decomposition for Multi-task Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Honguk Woo, Minjong Yoo, Sangwoo Cho","submitted_at":"2024-08-28T07:36:20Z","abstract_excerpt":"Reinforcement learning (RL) with diverse offline datasets can have the advantage of leveraging the relation of multiple tasks and the common skills learned across those tasks, hence allowing us to deal with real-world complex problems efficiently in a data-driven way. In offline RL where only offline data is used and online interaction with the environment is restricted, it is yet difficult to achieve the optimal policy for multiple tasks, especially when the data quality varies for the tasks. In this paper, we present a skill-based multi-task RL technique on heterogeneous datasets that are ge"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.15593","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.15593/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:00:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4s8nHp3CBv1SHdxluvL/8BF7FG+ZGSxl3fW953Wjf/xBh+VXa89p3TZGR5rYFLM7H+AHxQ32j4GWPGxFu/MlBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T06:27:18.727097Z"},"content_sha256":"f3e0402bcb11b781c96008cc640c767afbf08cde6f154696e2fcd42f24747c48","schema_version":"1.0","event_id":"sha256:f3e0402bcb11b781c96008cc640c767afbf08cde6f154696e2fcd42f24747c48"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GEZQSGYMFQO6RWRD7XNUWZT6I5/bundle.json","state_url":"https://pith.science/pith/GEZQSGYMFQO6RWRD7XNUWZT6I5/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GEZQSGYMFQO6RWRD7XNUWZT6I5/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-20T06:27:18Z","links":{"resolver":"https://pith.science/pith/GEZQSGYMFQO6RWRD7XNUWZT6I5","bundle":"https://pith.science/pith/GEZQSGYMFQO6RWRD7XNUWZT6I5/bundle.json","state":"https://pith.science/pith/GEZQSGYMFQO6RWRD7XNUWZT6I5/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GEZQSGYMFQO6RWRD7XNUWZT6I5/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:GEZQSGYMFQO6RWRD7XNUWZT6I5","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"54099a741ec3613027cdb8f26f68614408ce7fbd909105431cd25d25a112ff04","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-28T07:36:20Z","title_canon_sha256":"dfa16a5680c6f12258629318ad146d23d2c527f0734dfba939494fc8a21e439d"},"schema_version":"1.0","source":{"id":"2408.15593","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2408.15593","created_at":"2026-07-05T09:00:14Z"},{"alias_kind":"arxiv_version","alias_value":"2408.15593v1","created_at":"2026-07-05T09:00:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.15593","created_at":"2026-07-05T09:00:14Z"},{"alias_kind":"pith_short_12","alias_value":"GEZQSGYMFQO6","created_at":"2026-07-05T09:00:14Z"},{"alias_kind":"pith_short_16","alias_value":"GEZQSGYMFQO6RWRD","created_at":"2026-07-05T09:00:14Z"},{"alias_kind":"pith_short_8","alias_value":"GEZQSGYM","created_at":"2026-07-05T09:00:14Z"}],"graph_snapshots":[{"event_id":"sha256:f3e0402bcb11b781c96008cc640c767afbf08cde6f154696e2fcd42f24747c48","target":"graph","created_at":"2026-07-05T09:00:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2408.15593/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) with diverse offline datasets can have the advantage of leveraging the relation of multiple tasks and the common skills learned across those tasks, hence allowing us to deal with real-world complex problems efficiently in a data-driven way. In offline RL where only offline data is used and online interaction with the environment is restricted, it is yet difficult to achieve the optimal policy for multiple tasks, especially when the data quality varies for the tasks. In this paper, we present a skill-based multi-task RL technique on heterogeneous datasets that are ge","authors_text":"Honguk Woo, Minjong Yoo, Sangwoo Cho","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-28T07:36:20Z","title":"Skills Regularized Task Decomposition for Multi-task Offline Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.15593","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:25708fac50fbf67f1000a649cc49138c8ebbfa2542289ea01adadb01063199a3","target":"record","created_at":"2026-07-05T09:00:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"54099a741ec3613027cdb8f26f68614408ce7fbd909105431cd25d25a112ff04","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-28T07:36:20Z","title_canon_sha256":"dfa16a5680c6f12258629318ad146d23d2c527f0734dfba939494fc8a21e439d"},"schema_version":"1.0","source":{"id":"2408.15593","kind":"arxiv","version":1}},"canonical_sha256":"3133091b0c2c1de8da23fddb4b667e4776c6d97edfdda6795266e6610ed4c97c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3133091b0c2c1de8da23fddb4b667e4776c6d97edfdda6795266e6610ed4c97c","first_computed_at":"2026-07-05T09:00:14.257934Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:00:14.257934Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"fF1b1X6ddcWKCLXeKk7jpkXanKLINSz8hFyDsbx0lllu0LdWykADmNmFkqT+4VldaBcThtO7hHOs2oFErXhDBg==","signature_status":"signed_v1","signed_at":"2026-07-05T09:00:14.258426Z","signed_message":"canonical_sha256_bytes"},"source_id":"2408.15593","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:25708fac50fbf67f1000a649cc49138c8ebbfa2542289ea01adadb01063199a3","sha256:f3e0402bcb11b781c96008cc640c767afbf08cde6f154696e2fcd42f24747c48"],"state_sha256":"43176848d8cf6c9012071d30ebc95f4bb8a16344ec507e2487cf95d433b5b7ad"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"prky7oo/FwYnEz7W4afS0LMQBh7CZkO8Z0u2Oqrfv5n03QiLG5w8Rvziaepv2fEXhLvpUCT7QkFMz3L4odqzCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-20T06:27:18.730689Z","bundle_sha256":"f6dfbf430897e9c2ba3175a2a34d9ae81d384037315610f21fddb28638737cc7"}}