{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:2TVHEGTWGEFDDOURNPECM7EAEB","short_pith_number":"pith:2TVHEGTW","canonical_record":{"source":{"id":"2504.08772","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-03T07:11:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"439176d2bcd0ffda471f51c703ac57a9912c3ad6b282b8e906f84c5881ed9386","abstract_canon_sha256":"f3f4c2d195d3a1a9e4511190f935091a74f59c2b5f10b0ebed68716daf4e6fb3"},"schema_version":"1.0"},"canonical_sha256":"d4ea721a76310a31ba916bc8267c802041c3d23e60a8597b7d886a64714655b6","source":{"kind":"arxiv","id":"2504.08772","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.08772","created_at":"2026-07-05T10:47:59Z"},{"alias_kind":"arxiv_version","alias_value":"2504.08772v1","created_at":"2026-07-05T10:47:59Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.08772","created_at":"2026-07-05T10:47:59Z"},{"alias_kind":"pith_short_12","alias_value":"2TVHEGTWGEFD","created_at":"2026-07-05T10:47:59Z"},{"alias_kind":"pith_short_16","alias_value":"2TVHEGTWGEFDDOUR","created_at":"2026-07-05T10:47:59Z"},{"alias_kind":"pith_short_8","alias_value":"2TVHEGTW","created_at":"2026-07-05T10:47:59Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:2TVHEGTWGEFDDOURNPECM7EAEB","target":"record","payload":{"canonical_record":{"source":{"id":"2504.08772","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-03T07:11:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"439176d2bcd0ffda471f51c703ac57a9912c3ad6b282b8e906f84c5881ed9386","abstract_canon_sha256":"f3f4c2d195d3a1a9e4511190f935091a74f59c2b5f10b0ebed68716daf4e6fb3"},"schema_version":"1.0"},"canonical_sha256":"d4ea721a76310a31ba916bc8267c802041c3d23e60a8597b7d886a64714655b6","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:59.120397Z","signature_b64":"P8bHRN0zleWNTnrEv6EPHtfsN3wYRgwqq/d1gaftglIYbl6uO2KpgDb+cj/xHUoQk+3aBCZVDT6JN+vGHyXtCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d4ea721a76310a31ba916bc8267c802041c3d23e60a8597b7d886a64714655b6","last_reissued_at":"2026-07-05T10:47:59.119942Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:59.119942Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2504.08772","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:47:59Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"oxyPP0M67ut31oLwsuH9PUecWlo0+kU0W7xug/S+DG0Rf+vDATAxClYgHzbAeYkUbADy9bAfhqGgfXkou+5XBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T11:00:11.853168Z"},"content_sha256":"5ef7027cf11c5a76f11590367a79f9e734bb78bf1fffbba8aff80cca51f5b0b4","schema_version":"1.0","event_id":"sha256:5ef7027cf11c5a76f11590367a79f9e734bb78bf1fffbba8aff80cca51f5b0b4"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:2TVHEGTWGEFDDOURNPECM7EAEB","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reward Generation via Large Vision-Language Model in Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chang D. Yoo, Donghoon Lee, Tung M. Luu, Younghwan Lee","submitted_at":"2025-04-03T07:11:18Z","abstract_excerpt":"In offline reinforcement learning (RL), learning from fixed datasets presents a promising solution for domains where real-time interaction with the environment is expensive or risky. However, designing dense reward signals for offline dataset requires significant human effort and domain expertise. Reinforcement learning with human feedback (RLHF) has emerged as an alternative, but it remains costly due to the human-in-the-loop process, prompting interest in automated reward generation models. To address this, we propose Reward Generation via Large Vision-Language Models (RG-VLM), which leverag"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.08772","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.08772/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:47:59Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"TnXTwj8r8qpVJFI942Hcb3DE4gYEeWnuYBSsZIX/FWvcJK42uGFCr5n2ZFdkOFWtpdLoRmbVUzVIgPJMx5UKBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T11:00:11.854071Z"},"content_sha256":"479de3cd2814c5989f187341b6a7be98ce2123556cb6be029b0b2c2bcdbc547c","schema_version":"1.0","event_id":"sha256:479de3cd2814c5989f187341b6a7be98ce2123556cb6be029b0b2c2bcdbc547c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2TVHEGTWGEFDDOURNPECM7EAEB/bundle.json","state_url":"https://pith.science/pith/2TVHEGTWGEFDDOURNPECM7EAEB/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2TVHEGTWGEFDDOURNPECM7EAEB/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T11:00:11Z","links":{"resolver":"https://pith.science/pith/2TVHEGTWGEFDDOURNPECM7EAEB","bundle":"https://pith.science/pith/2TVHEGTWGEFDDOURNPECM7EAEB/bundle.json","state":"https://pith.science/pith/2TVHEGTWGEFDDOURNPECM7EAEB/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2TVHEGTWGEFDDOURNPECM7EAEB/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:2TVHEGTWGEFDDOURNPECM7EAEB","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f3f4c2d195d3a1a9e4511190f935091a74f59c2b5f10b0ebed68716daf4e6fb3","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-03T07:11:18Z","title_canon_sha256":"439176d2bcd0ffda471f51c703ac57a9912c3ad6b282b8e906f84c5881ed9386"},"schema_version":"1.0","source":{"id":"2504.08772","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.08772","created_at":"2026-07-05T10:47:59Z"},{"alias_kind":"arxiv_version","alias_value":"2504.08772v1","created_at":"2026-07-05T10:47:59Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.08772","created_at":"2026-07-05T10:47:59Z"},{"alias_kind":"pith_short_12","alias_value":"2TVHEGTWGEFD","created_at":"2026-07-05T10:47:59Z"},{"alias_kind":"pith_short_16","alias_value":"2TVHEGTWGEFDDOUR","created_at":"2026-07-05T10:47:59Z"},{"alias_kind":"pith_short_8","alias_value":"2TVHEGTW","created_at":"2026-07-05T10:47:59Z"}],"graph_snapshots":[{"event_id":"sha256:479de3cd2814c5989f187341b6a7be98ce2123556cb6be029b0b2c2bcdbc547c","target":"graph","created_at":"2026-07-05T10:47:59Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2504.08772/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In offline reinforcement learning (RL), learning from fixed datasets presents a promising solution for domains where real-time interaction with the environment is expensive or risky. However, designing dense reward signals for offline dataset requires significant human effort and domain expertise. Reinforcement learning with human feedback (RLHF) has emerged as an alternative, but it remains costly due to the human-in-the-loop process, prompting interest in automated reward generation models. To address this, we propose Reward Generation via Large Vision-Language Models (RG-VLM), which leverag","authors_text":"Chang D. Yoo, Donghoon Lee, Tung M. Luu, Younghwan Lee","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-03T07:11:18Z","title":"Reward Generation via Large Vision-Language Model in Offline Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.08772","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5ef7027cf11c5a76f11590367a79f9e734bb78bf1fffbba8aff80cca51f5b0b4","target":"record","created_at":"2026-07-05T10:47:59Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f3f4c2d195d3a1a9e4511190f935091a74f59c2b5f10b0ebed68716daf4e6fb3","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-03T07:11:18Z","title_canon_sha256":"439176d2bcd0ffda471f51c703ac57a9912c3ad6b282b8e906f84c5881ed9386"},"schema_version":"1.0","source":{"id":"2504.08772","kind":"arxiv","version":1}},"canonical_sha256":"d4ea721a76310a31ba916bc8267c802041c3d23e60a8597b7d886a64714655b6","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d4ea721a76310a31ba916bc8267c802041c3d23e60a8597b7d886a64714655b6","first_computed_at":"2026-07-05T10:47:59.119942Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:47:59.119942Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"P8bHRN0zleWNTnrEv6EPHtfsN3wYRgwqq/d1gaftglIYbl6uO2KpgDb+cj/xHUoQk+3aBCZVDT6JN+vGHyXtCg==","signature_status":"signed_v1","signed_at":"2026-07-05T10:47:59.120397Z","signed_message":"canonical_sha256_bytes"},"source_id":"2504.08772","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5ef7027cf11c5a76f11590367a79f9e734bb78bf1fffbba8aff80cca51f5b0b4","sha256:479de3cd2814c5989f187341b6a7be98ce2123556cb6be029b0b2c2bcdbc547c"],"state_sha256":"a4841be0a29447dc4f9b967200c93921cd9ed19530982574d96e93e90866e4e0"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"AT+GoFBjhquNSXCkeXzn6N1IMFeRICRtg3sTFVgAfbe8JLHRTbe9/2IVt7dGqDHF9OlpoGy8jS8iY5ftHwwpAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T11:00:11.861363Z","bundle_sha256":"8a9b5b8bb41f0520fe8562a4a888da4adf1cf65e050d39ba17c37316e94def40"}}