{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:KPVR2GU5HBUC7AJ7JJGCSHUYTT","short_pith_number":"pith:KPVR2GU5","canonical_record":{"source":{"id":"2502.00601","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-02T00:03:53Z","cross_cats_sorted":[],"title_canon_sha256":"20bbb304f0648770f7d5e4867a1558d80e78fd5139d14d24805cc6133c6f4b53","abstract_canon_sha256":"825ff9002e81b20f700542ce710f041cb8dddb5b3a7aa6aefaa3bb7bd03f7ebd"},"schema_version":"1.0"},"canonical_sha256":"53eb1d1a9d38682f813f4a4c291e989ce59b761961427ad824a8320936431741","source":{"kind":"arxiv","id":"2502.00601","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.00601","created_at":"2026-07-05T10:48:25Z"},{"alias_kind":"arxiv_version","alias_value":"2502.00601v2","created_at":"2026-07-05T10:48:25Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.00601","created_at":"2026-07-05T10:48:25Z"},{"alias_kind":"pith_short_12","alias_value":"KPVR2GU5HBUC","created_at":"2026-07-05T10:48:25Z"},{"alias_kind":"pith_short_16","alias_value":"KPVR2GU5HBUC7AJ7","created_at":"2026-07-05T10:48:25Z"},{"alias_kind":"pith_short_8","alias_value":"KPVR2GU5","created_at":"2026-07-05T10:48:25Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:KPVR2GU5HBUC7AJ7JJGCSHUYTT","target":"record","payload":{"canonical_record":{"source":{"id":"2502.00601","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-02T00:03:53Z","cross_cats_sorted":[],"title_canon_sha256":"20bbb304f0648770f7d5e4867a1558d80e78fd5139d14d24805cc6133c6f4b53","abstract_canon_sha256":"825ff9002e81b20f700542ce710f041cb8dddb5b3a7aa6aefaa3bb7bd03f7ebd"},"schema_version":"1.0"},"canonical_sha256":"53eb1d1a9d38682f813f4a4c291e989ce59b761961427ad824a8320936431741","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:48:25.030573Z","signature_b64":"bl4ODIr5oxrU4VhhnfjjVuvxazfrNOHth7oxarUo0PsHOyfg4VdMMv6VdZ/4JFxxWd580WVs+lUwxua8CKvzAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"53eb1d1a9d38682f813f4a4c291e989ce59b761961427ad824a8320936431741","last_reissued_at":"2026-07-05T10:48:25.030021Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:48:25.030021Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2502.00601","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:48:25Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"8G3B3oF7RmScsQekbqFf+nfv5kpWLwIdBhMuuYkT3dIjvPIimsr6GPCaLyepQZI5jQH4uItnC5kqhf5Nf61VBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T00:05:47.493236Z"},"content_sha256":"ed1d00c933730d0567a7abc555ccf6ba72364fa4200c188b45c78afa393172fc","schema_version":"1.0","event_id":"sha256:ed1d00c933730d0567a7abc555ccf6ba72364fa4200c188b45c78afa393172fc"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:KPVR2GU5HBUC7AJ7JJGCSHUYTT","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Enhancing Offline Reinforcement Learning with Curriculum Learning-Based Trajectory Valuation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amir Abolfazli, Avishek Anand, Wolfgang Nejdl, Zekun Song","submitted_at":"2025-02-02T00:03:53Z","abstract_excerpt":"The success of deep reinforcement learning (DRL) relies on the availability and quality of training data, often requiring extensive interactions with specific environments. In many real-world scenarios, where data collection is costly and risky, offline reinforcement learning (RL) offers a solution by utilizing data collected by domain experts and searching for a batch-constrained optimal policy. This approach is further augmented by incorporating external data sources, expanding the range and diversity of data collection possibilities. However, existing offline RL methods often struggle with "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.00601","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.00601/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:48:25Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"NU2NT7/OAc5Xj8Sz/EVif5r2mnlhZYlzVAWJYaSK1bLA5/lkonWZt8n3Eu3P8NVzLTAY1V1nTOQUHmCzEiyYAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T00:05:47.494108Z"},"content_sha256":"3519c16a7b8f8847080402a10f2c30ac21ceb6d94a322535a6c8d6757fed65d6","schema_version":"1.0","event_id":"sha256:3519c16a7b8f8847080402a10f2c30ac21ceb6d94a322535a6c8d6757fed65d6"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/KPVR2GU5HBUC7AJ7JJGCSHUYTT/bundle.json","state_url":"https://pith.science/pith/KPVR2GU5HBUC7AJ7JJGCSHUYTT/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/KPVR2GU5HBUC7AJ7JJGCSHUYTT/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-11T00:05:47Z","links":{"resolver":"https://pith.science/pith/KPVR2GU5HBUC7AJ7JJGCSHUYTT","bundle":"https://pith.science/pith/KPVR2GU5HBUC7AJ7JJGCSHUYTT/bundle.json","state":"https://pith.science/pith/KPVR2GU5HBUC7AJ7JJGCSHUYTT/state.json","well_known_bundle":"https://pith.science/.well-known/pith/KPVR2GU5HBUC7AJ7JJGCSHUYTT/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:KPVR2GU5HBUC7AJ7JJGCSHUYTT","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"825ff9002e81b20f700542ce710f041cb8dddb5b3a7aa6aefaa3bb7bd03f7ebd","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-02T00:03:53Z","title_canon_sha256":"20bbb304f0648770f7d5e4867a1558d80e78fd5139d14d24805cc6133c6f4b53"},"schema_version":"1.0","source":{"id":"2502.00601","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.00601","created_at":"2026-07-05T10:48:25Z"},{"alias_kind":"arxiv_version","alias_value":"2502.00601v2","created_at":"2026-07-05T10:48:25Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.00601","created_at":"2026-07-05T10:48:25Z"},{"alias_kind":"pith_short_12","alias_value":"KPVR2GU5HBUC","created_at":"2026-07-05T10:48:25Z"},{"alias_kind":"pith_short_16","alias_value":"KPVR2GU5HBUC7AJ7","created_at":"2026-07-05T10:48:25Z"},{"alias_kind":"pith_short_8","alias_value":"KPVR2GU5","created_at":"2026-07-05T10:48:25Z"}],"graph_snapshots":[{"event_id":"sha256:3519c16a7b8f8847080402a10f2c30ac21ceb6d94a322535a6c8d6757fed65d6","target":"graph","created_at":"2026-07-05T10:48:25Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.00601/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The success of deep reinforcement learning (DRL) relies on the availability and quality of training data, often requiring extensive interactions with specific environments. In many real-world scenarios, where data collection is costly and risky, offline reinforcement learning (RL) offers a solution by utilizing data collected by domain experts and searching for a batch-constrained optimal policy. This approach is further augmented by incorporating external data sources, expanding the range and diversity of data collection possibilities. However, existing offline RL methods often struggle with ","authors_text":"Amir Abolfazli, Avishek Anand, Wolfgang Nejdl, Zekun Song","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-02T00:03:53Z","title":"Enhancing Offline Reinforcement Learning with Curriculum Learning-Based Trajectory Valuation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.00601","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ed1d00c933730d0567a7abc555ccf6ba72364fa4200c188b45c78afa393172fc","target":"record","created_at":"2026-07-05T10:48:25Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"825ff9002e81b20f700542ce710f041cb8dddb5b3a7aa6aefaa3bb7bd03f7ebd","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-02T00:03:53Z","title_canon_sha256":"20bbb304f0648770f7d5e4867a1558d80e78fd5139d14d24805cc6133c6f4b53"},"schema_version":"1.0","source":{"id":"2502.00601","kind":"arxiv","version":2}},"canonical_sha256":"53eb1d1a9d38682f813f4a4c291e989ce59b761961427ad824a8320936431741","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"53eb1d1a9d38682f813f4a4c291e989ce59b761961427ad824a8320936431741","first_computed_at":"2026-07-05T10:48:25.030021Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:48:25.030021Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"bl4ODIr5oxrU4VhhnfjjVuvxazfrNOHth7oxarUo0PsHOyfg4VdMMv6VdZ/4JFxxWd580WVs+lUwxua8CKvzAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T10:48:25.030573Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.00601","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ed1d00c933730d0567a7abc555ccf6ba72364fa4200c188b45c78afa393172fc","sha256:3519c16a7b8f8847080402a10f2c30ac21ceb6d94a322535a6c8d6757fed65d6"],"state_sha256":"09b14c9325f66d9b3df5786650e10b1ec2ebd5e97531054daf2e41d7ab5d869e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+aFWKi8E+RPv1uHyg7oLTXYQK4L/20b1IKpjyytITNOu6jU9DTOmdg+EpKospXHfgkrbcJedDVkT/fyKWab+AA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-11T00:05:47.499722Z","bundle_sha256":"4299a4c356a8ad539587d3f37dd4bf8bccbe0044881ff255bf675c6126b79bc4"}}