{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:YQCAERWFT5ZBROC6ODXHIBJA5Z","short_pith_number":"pith:YQCAERWF","canonical_record":{"source":{"id":"2509.00347","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-30T04:02:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e8539c4e122a4775bd1f3f2a2acfc5860a3815f119ae8a5e56626a6787497c0b","abstract_canon_sha256":"926132d91c59bcf8efa47a92668f20c6b1e868716df1b37888db3e2bdf3e13c3"},"schema_version":"1.0"},"canonical_sha256":"c4040246c59f7218b85e70ee740520ee4686aaabf47f4eee5330ee9a0272855f","source":{"kind":"arxiv","id":"2509.00347","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2509.00347","created_at":"2026-07-05T12:02:14Z"},{"alias_kind":"arxiv_version","alias_value":"2509.00347v1","created_at":"2026-07-05T12:02:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.00347","created_at":"2026-07-05T12:02:14Z"},{"alias_kind":"pith_short_12","alias_value":"YQCAERWFT5ZB","created_at":"2026-07-05T12:02:14Z"},{"alias_kind":"pith_short_16","alias_value":"YQCAERWFT5ZBROC6","created_at":"2026-07-05T12:02:14Z"},{"alias_kind":"pith_short_8","alias_value":"YQCAERWF","created_at":"2026-07-05T12:02:14Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:YQCAERWFT5ZBROC6ODXHIBJA5Z","target":"record","payload":{"canonical_record":{"source":{"id":"2509.00347","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-30T04:02:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e8539c4e122a4775bd1f3f2a2acfc5860a3815f119ae8a5e56626a6787497c0b","abstract_canon_sha256":"926132d91c59bcf8efa47a92668f20c6b1e868716df1b37888db3e2bdf3e13c3"},"schema_version":"1.0"},"canonical_sha256":"c4040246c59f7218b85e70ee740520ee4686aaabf47f4eee5330ee9a0272855f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:02:14.202960Z","signature_b64":"xNkPGmh06XsnuvqyzzjMQusFRZDftM2tL0o1gmO0wR5vQqfhL6ppoAX842wSsrxPHJV5DskKx2cy5tcIWIE1BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4040246c59f7218b85e70ee740520ee4686aaabf47f4eee5330ee9a0272855f","last_reissued_at":"2026-07-05T12:02:14.202479Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:02:14.202479Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2509.00347","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:02:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"L7+UxupyPUpcuwzrjRj4zJ/avVXrRitPbFSbO1I/WnOrbu2dhvWgVKLhAdVs1GTCafOS8h3WRQ+NlQkBU7m2Dg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T03:09:56.094551Z"},"content_sha256":"713edf1ff665c95e46dbf95dccf37443e1a127825bc2246ad45589395279bab0","schema_version":"1.0","event_id":"sha256:713edf1ff665c95e46dbf95dccf37443e1a127825bc2246ad45589395279bab0"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:YQCAERWFT5ZBROC6ODXHIBJA5Z","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"LLM-Driven Policy Diffusion: Enhancing Generalization in Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Hanping Zhang, Yuhong Guo","submitted_at":"2025-08-30T04:02:33Z","abstract_excerpt":"Reinforcement Learning (RL) is known for its strong decision-making capabilities and has been widely applied in various real-world scenarios. However, with the increasing availability of offline datasets and the lack of well-designed online environments from human experts, the challenge of generalization in offline RL has become more prominent. Due to the limitations of offline data, RL agents trained solely on collected experiences often struggle to generalize to new tasks or environments. To address this challenge, we propose LLM-Driven Policy Diffusion (LLMDPD), a novel approach that enhanc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.00347","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.00347/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:02:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xR4a5uHdp5iiQy1Yxl76kY+YuJMX3C9aVFUT1vF68tSJXG+MTon7Wx+EL6fHIlEjwnBYQQMYvA8S6XjsW6DGBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T03:09:56.095127Z"},"content_sha256":"4cac72f70fe05cf11acdac80e6e6bbfcc5d0d3947aec3ac77d7492380080a31d","schema_version":"1.0","event_id":"sha256:4cac72f70fe05cf11acdac80e6e6bbfcc5d0d3947aec3ac77d7492380080a31d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/YQCAERWFT5ZBROC6ODXHIBJA5Z/bundle.json","state_url":"https://pith.science/pith/YQCAERWFT5ZBROC6ODXHIBJA5Z/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/YQCAERWFT5ZBROC6ODXHIBJA5Z/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T03:09:56Z","links":{"resolver":"https://pith.science/pith/YQCAERWFT5ZBROC6ODXHIBJA5Z","bundle":"https://pith.science/pith/YQCAERWFT5ZBROC6ODXHIBJA5Z/bundle.json","state":"https://pith.science/pith/YQCAERWFT5ZBROC6ODXHIBJA5Z/state.json","well_known_bundle":"https://pith.science/.well-known/pith/YQCAERWFT5ZBROC6ODXHIBJA5Z/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:YQCAERWFT5ZBROC6ODXHIBJA5Z","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"926132d91c59bcf8efa47a92668f20c6b1e868716df1b37888db3e2bdf3e13c3","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-30T04:02:33Z","title_canon_sha256":"e8539c4e122a4775bd1f3f2a2acfc5860a3815f119ae8a5e56626a6787497c0b"},"schema_version":"1.0","source":{"id":"2509.00347","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2509.00347","created_at":"2026-07-05T12:02:14Z"},{"alias_kind":"arxiv_version","alias_value":"2509.00347v1","created_at":"2026-07-05T12:02:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.00347","created_at":"2026-07-05T12:02:14Z"},{"alias_kind":"pith_short_12","alias_value":"YQCAERWFT5ZB","created_at":"2026-07-05T12:02:14Z"},{"alias_kind":"pith_short_16","alias_value":"YQCAERWFT5ZBROC6","created_at":"2026-07-05T12:02:14Z"},{"alias_kind":"pith_short_8","alias_value":"YQCAERWF","created_at":"2026-07-05T12:02:14Z"}],"graph_snapshots":[{"event_id":"sha256:4cac72f70fe05cf11acdac80e6e6bbfcc5d0d3947aec3ac77d7492380080a31d","target":"graph","created_at":"2026-07-05T12:02:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2509.00347/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning (RL) is known for its strong decision-making capabilities and has been widely applied in various real-world scenarios. However, with the increasing availability of offline datasets and the lack of well-designed online environments from human experts, the challenge of generalization in offline RL has become more prominent. Due to the limitations of offline data, RL agents trained solely on collected experiences often struggle to generalize to new tasks or environments. To address this challenge, we propose LLM-Driven Policy Diffusion (LLMDPD), a novel approach that enhanc","authors_text":"Hanping Zhang, Yuhong Guo","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-30T04:02:33Z","title":"LLM-Driven Policy Diffusion: Enhancing Generalization in Offline Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.00347","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:713edf1ff665c95e46dbf95dccf37443e1a127825bc2246ad45589395279bab0","target":"record","created_at":"2026-07-05T12:02:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"926132d91c59bcf8efa47a92668f20c6b1e868716df1b37888db3e2bdf3e13c3","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-30T04:02:33Z","title_canon_sha256":"e8539c4e122a4775bd1f3f2a2acfc5860a3815f119ae8a5e56626a6787497c0b"},"schema_version":"1.0","source":{"id":"2509.00347","kind":"arxiv","version":1}},"canonical_sha256":"c4040246c59f7218b85e70ee740520ee4686aaabf47f4eee5330ee9a0272855f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"c4040246c59f7218b85e70ee740520ee4686aaabf47f4eee5330ee9a0272855f","first_computed_at":"2026-07-05T12:02:14.202479Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T12:02:14.202479Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"xNkPGmh06XsnuvqyzzjMQusFRZDftM2tL0o1gmO0wR5vQqfhL6ppoAX842wSsrxPHJV5DskKx2cy5tcIWIE1BA==","signature_status":"signed_v1","signed_at":"2026-07-05T12:02:14.202960Z","signed_message":"canonical_sha256_bytes"},"source_id":"2509.00347","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:713edf1ff665c95e46dbf95dccf37443e1a127825bc2246ad45589395279bab0","sha256:4cac72f70fe05cf11acdac80e6e6bbfcc5d0d3947aec3ac77d7492380080a31d"],"state_sha256":"e6d0069ab51f4fa4ad852181295a4975c0df44ced65a17744ff9121abdd765c5"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"sfPtX5qZ09fV/AzCnNjr8LVHQKgqGitDsUILyJJ45NA0+WFY4I12YpFRiN8fRq0ljhn3KzE0OIr6PBpWWdi4BA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T03:09:56.100290Z","bundle_sha256":"483fa4926a969aebbe9fc09aaba6986863b78c4451551ec805e5ae660c3aadd3"}}