{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:I74ZOJGMMUES7EJNGBL4AVB6I4","short_pith_number":"pith:I74ZOJGM","canonical_record":{"source":{"id":"2402.04764","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-07T11:27:45Z","cross_cats_sorted":[],"title_canon_sha256":"afdb49aa42eb975a3020db973f91ba70ab32c9ce395a67a4d8245c705ca95439","abstract_canon_sha256":"7c752f9910a63a8c17da1336d6fef3aa25db9703efa161513a4086d940292573"},"schema_version":"1.0"},"canonical_sha256":"47f99724cc65092f912d3057c0543e47337ac1135d693e4fcfd07aa6e108eda9","source":{"kind":"arxiv","id":"2402.04764","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2402.04764","created_at":"2026-07-05T07:42:32Z"},{"alias_kind":"arxiv_version","alias_value":"2402.04764v1","created_at":"2026-07-05T07:42:32Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.04764","created_at":"2026-07-05T07:42:32Z"},{"alias_kind":"pith_short_12","alias_value":"I74ZOJGMMUES","created_at":"2026-07-05T07:42:32Z"},{"alias_kind":"pith_short_16","alias_value":"I74ZOJGMMUES7EJN","created_at":"2026-07-05T07:42:32Z"},{"alias_kind":"pith_short_8","alias_value":"I74ZOJGM","created_at":"2026-07-05T07:42:32Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:I74ZOJGMMUES7EJNGBL4AVB6I4","target":"record","payload":{"canonical_record":{"source":{"id":"2402.04764","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-07T11:27:45Z","cross_cats_sorted":[],"title_canon_sha256":"afdb49aa42eb975a3020db973f91ba70ab32c9ce395a67a4d8245c705ca95439","abstract_canon_sha256":"7c752f9910a63a8c17da1336d6fef3aa25db9703efa161513a4086d940292573"},"schema_version":"1.0"},"canonical_sha256":"47f99724cc65092f912d3057c0543e47337ac1135d693e4fcfd07aa6e108eda9","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:42:32.118574Z","signature_b64":"XctPLt4v2FeB46V2vhOuXNvGM9lXkjMDGB5OfYzt8mn2tRRb2tA3a6f/tHvwCAOKkfugegpCmJUmF1LlbSU6DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"47f99724cc65092f912d3057c0543e47337ac1135d693e4fcfd07aa6e108eda9","last_reissued_at":"2026-07-05T07:42:32.118140Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:42:32.118140Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2402.04764","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:42:32Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"N1a1N+nwgh2gZ2FyIxOFY8EM6vk2NvMq47drIoxhbmP66NCL2v408KofubIuKTW06NPJYVaaduRcyUFVDwxMCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T18:51:38.234521Z"},"content_sha256":"b52feafb3783f83a04098be1712c63d6103bdd9e98f20166570ed1da274b4740","schema_version":"1.0","event_id":"sha256:b52feafb3783f83a04098be1712c63d6103bdd9e98f20166570ed1da274b4740"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:I74ZOJGMMUES7EJNGBL4AVB6I4","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Code as Reward: Empowering Reinforcement Learning with VLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ankit Anand, David Venuto, Doina Precup, Martin Klissarov, Sami Nur Islam, Sherry Yang","submitted_at":"2024-02-07T11:27:45Z","abstract_excerpt":"Pre-trained Vision-Language Models (VLMs) are able to understand visual concepts, describe and decompose complex tasks into sub-tasks, and provide feedback on task completion. In this paper, we aim to leverage these capabilities to support the training of reinforcement learning (RL) agents. In principle, VLMs are well suited for this purpose, as they can naturally analyze image-based observations and provide feedback (reward) on learning progress. However, inference in VLMs is computationally expensive, so querying them frequently to compute rewards would significantly slowdown the training of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.04764","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.04764/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:42:32Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0KIyOD774JSS/IuWWRvtKozlmvt71H9worz+LCdajl8BjWtgYD5ZdF2IyMwuCkD25YE6INlAE1uYKDnwNpcrAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T18:51:38.235062Z"},"content_sha256":"4acdecdde2334fc35cf0002e20a1469fd3ba3f7551a6cfc05db5b21d30fe8986","schema_version":"1.0","event_id":"sha256:4acdecdde2334fc35cf0002e20a1469fd3ba3f7551a6cfc05db5b21d30fe8986"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/I74ZOJGMMUES7EJNGBL4AVB6I4/bundle.json","state_url":"https://pith.science/pith/I74ZOJGMMUES7EJNGBL4AVB6I4/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/I74ZOJGMMUES7EJNGBL4AVB6I4/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T18:51:38Z","links":{"resolver":"https://pith.science/pith/I74ZOJGMMUES7EJNGBL4AVB6I4","bundle":"https://pith.science/pith/I74ZOJGMMUES7EJNGBL4AVB6I4/bundle.json","state":"https://pith.science/pith/I74ZOJGMMUES7EJNGBL4AVB6I4/state.json","well_known_bundle":"https://pith.science/.well-known/pith/I74ZOJGMMUES7EJNGBL4AVB6I4/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:I74ZOJGMMUES7EJNGBL4AVB6I4","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"7c752f9910a63a8c17da1336d6fef3aa25db9703efa161513a4086d940292573","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-07T11:27:45Z","title_canon_sha256":"afdb49aa42eb975a3020db973f91ba70ab32c9ce395a67a4d8245c705ca95439"},"schema_version":"1.0","source":{"id":"2402.04764","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2402.04764","created_at":"2026-07-05T07:42:32Z"},{"alias_kind":"arxiv_version","alias_value":"2402.04764v1","created_at":"2026-07-05T07:42:32Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.04764","created_at":"2026-07-05T07:42:32Z"},{"alias_kind":"pith_short_12","alias_value":"I74ZOJGMMUES","created_at":"2026-07-05T07:42:32Z"},{"alias_kind":"pith_short_16","alias_value":"I74ZOJGMMUES7EJN","created_at":"2026-07-05T07:42:32Z"},{"alias_kind":"pith_short_8","alias_value":"I74ZOJGM","created_at":"2026-07-05T07:42:32Z"}],"graph_snapshots":[{"event_id":"sha256:4acdecdde2334fc35cf0002e20a1469fd3ba3f7551a6cfc05db5b21d30fe8986","target":"graph","created_at":"2026-07-05T07:42:32Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2402.04764/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Pre-trained Vision-Language Models (VLMs) are able to understand visual concepts, describe and decompose complex tasks into sub-tasks, and provide feedback on task completion. In this paper, we aim to leverage these capabilities to support the training of reinforcement learning (RL) agents. In principle, VLMs are well suited for this purpose, as they can naturally analyze image-based observations and provide feedback (reward) on learning progress. However, inference in VLMs is computationally expensive, so querying them frequently to compute rewards would significantly slowdown the training of","authors_text":"Ankit Anand, David Venuto, Doina Precup, Martin Klissarov, Sami Nur Islam, Sherry Yang","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-07T11:27:45Z","title":"Code as Reward: Empowering Reinforcement Learning with VLMs"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.04764","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b52feafb3783f83a04098be1712c63d6103bdd9e98f20166570ed1da274b4740","target":"record","created_at":"2026-07-05T07:42:32Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"7c752f9910a63a8c17da1336d6fef3aa25db9703efa161513a4086d940292573","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-07T11:27:45Z","title_canon_sha256":"afdb49aa42eb975a3020db973f91ba70ab32c9ce395a67a4d8245c705ca95439"},"schema_version":"1.0","source":{"id":"2402.04764","kind":"arxiv","version":1}},"canonical_sha256":"47f99724cc65092f912d3057c0543e47337ac1135d693e4fcfd07aa6e108eda9","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"47f99724cc65092f912d3057c0543e47337ac1135d693e4fcfd07aa6e108eda9","first_computed_at":"2026-07-05T07:42:32.118140Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:42:32.118140Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"XctPLt4v2FeB46V2vhOuXNvGM9lXkjMDGB5OfYzt8mn2tRRb2tA3a6f/tHvwCAOKkfugegpCmJUmF1LlbSU6DQ==","signature_status":"signed_v1","signed_at":"2026-07-05T07:42:32.118574Z","signed_message":"canonical_sha256_bytes"},"source_id":"2402.04764","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b52feafb3783f83a04098be1712c63d6103bdd9e98f20166570ed1da274b4740","sha256:4acdecdde2334fc35cf0002e20a1469fd3ba3f7551a6cfc05db5b21d30fe8986"],"state_sha256":"576ffbe1b991adb35678d56a11ae035f20e7a7718418a9ac0de1a322ef93470a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"TwpBH2HQXX8fibPBRvV/vz8kDqDYmY0M5m3ylJhZwW3NVEyW5GLnpUbHPmb6QbfQvxXCLaiJUH8MkbSnFg/tAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T18:51:38.238808Z","bundle_sha256":"2732ac9bec574342290accc7d73fb9f1cfa7a76877f45374deaae9c9c22f5e64"}}