{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:H7V2JPSAOBZDRYQ4HYHIPQQ4K7","short_pith_number":"pith:H7V2JPSA","canonical_record":{"source":{"id":"2607.01721","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2026-07-02T05:20:47Z","cross_cats_sorted":[],"title_canon_sha256":"5f6b4fa431f1b74532c380b8795d2bee558c78bed7ec43baa18fe8d05a885227","abstract_canon_sha256":"9f0418d6fe1e8b8799ee6d7fa7e4faa1ad54d83407a261d25cfcfbe56168cd56"},"schema_version":"1.0"},"canonical_sha256":"3feba4be40707238e21c3e0e87c21c57e36708721e2173bce87dfc9841996ee0","source":{"kind":"arxiv","id":"2607.01721","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.01721","created_at":"2026-07-03T01:17:27Z"},{"alias_kind":"arxiv_version","alias_value":"2607.01721v1","created_at":"2026-07-03T01:17:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.01721","created_at":"2026-07-03T01:17:27Z"},{"alias_kind":"pith_short_12","alias_value":"H7V2JPSAOBZD","created_at":"2026-07-03T01:17:27Z"},{"alias_kind":"pith_short_16","alias_value":"H7V2JPSAOBZDRYQ4","created_at":"2026-07-03T01:17:27Z"},{"alias_kind":"pith_short_8","alias_value":"H7V2JPSA","created_at":"2026-07-03T01:17:27Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:H7V2JPSAOBZDRYQ4HYHIPQQ4K7","target":"record","payload":{"canonical_record":{"source":{"id":"2607.01721","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2026-07-02T05:20:47Z","cross_cats_sorted":[],"title_canon_sha256":"5f6b4fa431f1b74532c380b8795d2bee558c78bed7ec43baa18fe8d05a885227","abstract_canon_sha256":"9f0418d6fe1e8b8799ee6d7fa7e4faa1ad54d83407a261d25cfcfbe56168cd56"},"schema_version":"1.0"},"canonical_sha256":"3feba4be40707238e21c3e0e87c21c57e36708721e2173bce87dfc9841996ee0","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-03T01:17:27.795727Z","signature_b64":"62SVpwE86E40Wsh8sOHWYKTqSzixISHwEbQ+RNFOhmmJRwaUO+qvk9R5b5SXs/AfLD8XSSGP58Dm6EVYhXd2Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3feba4be40707238e21c3e0e87c21c57e36708721e2173bce87dfc9841996ee0","last_reissued_at":"2026-07-03T01:17:27.795304Z","signature_status":"signed_v1","first_computed_at":"2026-07-03T01:17:27.795304Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.01721","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-03T01:17:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4NQRq+pVQ68DzwfYTLCC0EeJcc9ZcXIq/B7jfptgru/QbEktn/gKXiPj+boWk9IIpaLcy5xr7jnlIJcCPXooBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-04T19:44:01.674743Z"},"content_sha256":"a9b119d6482116ee6fe3b2ff82dc9e511468a17563e095ca241db1ac34655fac","schema_version":"1.0","event_id":"sha256:a9b119d6482116ee6fe3b2ff82dc9e511468a17563e095ca241db1ac34655fac"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:H7V2JPSAOBZDRYQ4HYHIPQQ4K7","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"CoRe: Combined Rewards with Vision-Language Model Feedback for Preference-Aligned Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Hexian Ni, Tao Lu, Yinghao Cai","submitted_at":"2026-07-02T05:20:47Z","abstract_excerpt":"Reward design remains a central challenge in reinforcement learning (RL). Hand-crafted rewards are often difficult to specify and may lead to suboptimal policies, while learned rewards from preferences can suffer from inefficiency and unstable training. Inspired by the dual nature of human learning explored in cognitive science, we decompose rewards into two complementary components: Formal Rewards (FR), explicitly designed based on task knowledge, and Residual Rewards (RR), learned from observations to capture implicit and nuanced preferences. Based on this decomposition, we propose CoRe, a h"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.01721","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.01721/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-03T01:17:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zNIwJfvRczySZGBFeA4W9Izju9DnLgfwepCa9eLxgnVL9l2dXGAs0RXNjgLSSFv4o2yut0j9OP5s+dMSPlFACA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-04T19:44:01.675145Z"},"content_sha256":"e20b4fff233a12790dbba52665fb9e4cde78c0377880bdabf2501a78bd817f46","schema_version":"1.0","event_id":"sha256:e20b4fff233a12790dbba52665fb9e4cde78c0377880bdabf2501a78bd817f46"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/H7V2JPSAOBZDRYQ4HYHIPQQ4K7/bundle.json","state_url":"https://pith.science/pith/H7V2JPSAOBZDRYQ4HYHIPQQ4K7/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/H7V2JPSAOBZDRYQ4HYHIPQQ4K7/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-04T19:44:01Z","links":{"resolver":"https://pith.science/pith/H7V2JPSAOBZDRYQ4HYHIPQQ4K7","bundle":"https://pith.science/pith/H7V2JPSAOBZDRYQ4HYHIPQQ4K7/bundle.json","state":"https://pith.science/pith/H7V2JPSAOBZDRYQ4HYHIPQQ4K7/state.json","well_known_bundle":"https://pith.science/.well-known/pith/H7V2JPSAOBZDRYQ4HYHIPQQ4K7/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:H7V2JPSAOBZDRYQ4HYHIPQQ4K7","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"9f0418d6fe1e8b8799ee6d7fa7e4faa1ad54d83407a261d25cfcfbe56168cd56","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2026-07-02T05:20:47Z","title_canon_sha256":"5f6b4fa431f1b74532c380b8795d2bee558c78bed7ec43baa18fe8d05a885227"},"schema_version":"1.0","source":{"id":"2607.01721","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.01721","created_at":"2026-07-03T01:17:27Z"},{"alias_kind":"arxiv_version","alias_value":"2607.01721v1","created_at":"2026-07-03T01:17:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.01721","created_at":"2026-07-03T01:17:27Z"},{"alias_kind":"pith_short_12","alias_value":"H7V2JPSAOBZD","created_at":"2026-07-03T01:17:27Z"},{"alias_kind":"pith_short_16","alias_value":"H7V2JPSAOBZDRYQ4","created_at":"2026-07-03T01:17:27Z"},{"alias_kind":"pith_short_8","alias_value":"H7V2JPSA","created_at":"2026-07-03T01:17:27Z"}],"graph_snapshots":[{"event_id":"sha256:e20b4fff233a12790dbba52665fb9e4cde78c0377880bdabf2501a78bd817f46","target":"graph","created_at":"2026-07-03T01:17:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.01721/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reward design remains a central challenge in reinforcement learning (RL). Hand-crafted rewards are often difficult to specify and may lead to suboptimal policies, while learned rewards from preferences can suffer from inefficiency and unstable training. Inspired by the dual nature of human learning explored in cognitive science, we decompose rewards into two complementary components: Formal Rewards (FR), explicitly designed based on task knowledge, and Residual Rewards (RR), learned from observations to capture implicit and nuanced preferences. Based on this decomposition, we propose CoRe, a h","authors_text":"Hexian Ni, Tao Lu, Yinghao Cai","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2026-07-02T05:20:47Z","title":"CoRe: Combined Rewards with Vision-Language Model Feedback for Preference-Aligned Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.01721","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:a9b119d6482116ee6fe3b2ff82dc9e511468a17563e095ca241db1ac34655fac","target":"record","created_at":"2026-07-03T01:17:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"9f0418d6fe1e8b8799ee6d7fa7e4faa1ad54d83407a261d25cfcfbe56168cd56","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2026-07-02T05:20:47Z","title_canon_sha256":"5f6b4fa431f1b74532c380b8795d2bee558c78bed7ec43baa18fe8d05a885227"},"schema_version":"1.0","source":{"id":"2607.01721","kind":"arxiv","version":1}},"canonical_sha256":"3feba4be40707238e21c3e0e87c21c57e36708721e2173bce87dfc9841996ee0","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3feba4be40707238e21c3e0e87c21c57e36708721e2173bce87dfc9841996ee0","first_computed_at":"2026-07-03T01:17:27.795304Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-03T01:17:27.795304Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"62SVpwE86E40Wsh8sOHWYKTqSzixISHwEbQ+RNFOhmmJRwaUO+qvk9R5b5SXs/AfLD8XSSGP58Dm6EVYhXd2Cw==","signature_status":"signed_v1","signed_at":"2026-07-03T01:17:27.795727Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.01721","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:a9b119d6482116ee6fe3b2ff82dc9e511468a17563e095ca241db1ac34655fac","sha256:e20b4fff233a12790dbba52665fb9e4cde78c0377880bdabf2501a78bd817f46"],"state_sha256":"66a226a5840421e1ea0f9b4328a60a043fb1fb3fd6313a8c44ed54f37928e46f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"tIHJMRICWMQU0PgQj4RRart8DOR64B5xXJ51kxehabarXFKiqVT0LWPZmBL9SbWLnbh+vtpC2etkqb3o8S+GCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-04T19:44:01.677293Z","bundle_sha256":"8d2e5d621f104587f242654936cdba16a677e6132af51b34ca12a9efd6fea199"}}