{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:PHQQ6W26XOY2D2BL6J7QWSQBJL","short_pith_number":"pith:PHQQ6W26","canonical_record":{"source":{"id":"2411.01111","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-02T02:22:21Z","cross_cats_sorted":[],"title_canon_sha256":"712aa99b70285964f2d87066f20802020e775d1d71da8e75bbed74f37a41a0b0","abstract_canon_sha256":"05d6fc19b4dea9acdd4ebc65022ee98a927820abea0590b69560ca004eb879bc"},"schema_version":"1.0"},"canonical_sha256":"79e10f5b5ebbb1a1e82bf27f0b4a014af6b4a96aa66c6186f0d516fdeec7bf21","source":{"kind":"arxiv","id":"2411.01111","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2411.01111","created_at":"2026-07-05T09:30:24Z"},{"alias_kind":"arxiv_version","alias_value":"2411.01111v1","created_at":"2026-07-05T09:30:24Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.01111","created_at":"2026-07-05T09:30:24Z"},{"alias_kind":"pith_short_12","alias_value":"PHQQ6W26XOY2","created_at":"2026-07-05T09:30:24Z"},{"alias_kind":"pith_short_16","alias_value":"PHQQ6W26XOY2D2BL","created_at":"2026-07-05T09:30:24Z"},{"alias_kind":"pith_short_8","alias_value":"PHQQ6W26","created_at":"2026-07-05T09:30:24Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:PHQQ6W26XOY2D2BL6J7QWSQBJL","target":"record","payload":{"canonical_record":{"source":{"id":"2411.01111","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-02T02:22:21Z","cross_cats_sorted":[],"title_canon_sha256":"712aa99b70285964f2d87066f20802020e775d1d71da8e75bbed74f37a41a0b0","abstract_canon_sha256":"05d6fc19b4dea9acdd4ebc65022ee98a927820abea0590b69560ca004eb879bc"},"schema_version":"1.0"},"canonical_sha256":"79e10f5b5ebbb1a1e82bf27f0b4a014af6b4a96aa66c6186f0d516fdeec7bf21","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:30:24.868545Z","signature_b64":"yXzOoaobztNejStOg7zdzXT/OiDBJxnEVtwpWKJyH4wg92ImI88naPVUNNBv12MXFSkYkhpaNFBxxs2WJuCUBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"79e10f5b5ebbb1a1e82bf27f0b4a014af6b4a96aa66c6186f0d516fdeec7bf21","last_reissued_at":"2026-07-05T09:30:24.868031Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:30:24.868031Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2411.01111","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:30:24Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"LC6H3Uya9Aa9s05mdOJCKrgn93CVujMyp11DRcY0bTcVd9L5YZDA0ZTHvAB0rH1pgsnlhl3gGfEX/09kDYoRBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T11:02:30.103444Z"},"content_sha256":"fd038064c3323f705dda3eb0d31d8bb926c7ac4a8266d0b86ed990184c15bf84","schema_version":"1.0","event_id":"sha256:fd038064c3323f705dda3eb0d31d8bb926c7ac4a8266d0b86ed990184c15bf84"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:PHQQ6W26XOY2D2BL6J7QWSQBJL","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Rule Based Rewards for Language Model Safety","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alec Helyar, Alex Beutel, Andrea Vallone, Ian Kivlichan, Johannes Heidecke, John Schulman, Joshua Achiam, Lilian Weng, Molly Lin, Tong Mu","submitted_at":"2024-11-02T02:22:21Z","abstract_excerpt":"Reinforcement learning based fine-tuning of large language models (LLMs) on human preferences has been shown to enhance both their capabilities and safety behavior. However, in cases related to safety, without precise instructions to human annotators, the data collected may cause the model to become overly cautious, or to respond in an undesirable style, such as being judgmental. Additionally, as model capabilities and usage patterns evolve, there may be a costly need to add or relabel data to modify safety behavior. We propose a novel preference modeling approach that utilizes AI feedback and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.01111","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.01111/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:30:24Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"e50YDb53Is7hmD3EkD2X66oVCYRTbFuSyaOOuNKadDhrgjX+AnB644FNLlSnlhY02FGnC38wGcndrr/vgiluDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T11:02:30.104300Z"},"content_sha256":"24bbddb3daff6ded2b60ee7c31c4df6083c73011145bcdc73dfd81846528e346","schema_version":"1.0","event_id":"sha256:24bbddb3daff6ded2b60ee7c31c4df6083c73011145bcdc73dfd81846528e346"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/PHQQ6W26XOY2D2BL6J7QWSQBJL/bundle.json","state_url":"https://pith.science/pith/PHQQ6W26XOY2D2BL6J7QWSQBJL/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/PHQQ6W26XOY2D2BL6J7QWSQBJL/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-01T11:02:30Z","links":{"resolver":"https://pith.science/pith/PHQQ6W26XOY2D2BL6J7QWSQBJL","bundle":"https://pith.science/pith/PHQQ6W26XOY2D2BL6J7QWSQBJL/bundle.json","state":"https://pith.science/pith/PHQQ6W26XOY2D2BL6J7QWSQBJL/state.json","well_known_bundle":"https://pith.science/.well-known/pith/PHQQ6W26XOY2D2BL6J7QWSQBJL/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:PHQQ6W26XOY2D2BL6J7QWSQBJL","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"05d6fc19b4dea9acdd4ebc65022ee98a927820abea0590b69560ca004eb879bc","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-02T02:22:21Z","title_canon_sha256":"712aa99b70285964f2d87066f20802020e775d1d71da8e75bbed74f37a41a0b0"},"schema_version":"1.0","source":{"id":"2411.01111","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2411.01111","created_at":"2026-07-05T09:30:24Z"},{"alias_kind":"arxiv_version","alias_value":"2411.01111v1","created_at":"2026-07-05T09:30:24Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.01111","created_at":"2026-07-05T09:30:24Z"},{"alias_kind":"pith_short_12","alias_value":"PHQQ6W26XOY2","created_at":"2026-07-05T09:30:24Z"},{"alias_kind":"pith_short_16","alias_value":"PHQQ6W26XOY2D2BL","created_at":"2026-07-05T09:30:24Z"},{"alias_kind":"pith_short_8","alias_value":"PHQQ6W26","created_at":"2026-07-05T09:30:24Z"}],"graph_snapshots":[{"event_id":"sha256:24bbddb3daff6ded2b60ee7c31c4df6083c73011145bcdc73dfd81846528e346","target":"graph","created_at":"2026-07-05T09:30:24Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2411.01111/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning based fine-tuning of large language models (LLMs) on human preferences has been shown to enhance both their capabilities and safety behavior. However, in cases related to safety, without precise instructions to human annotators, the data collected may cause the model to become overly cautious, or to respond in an undesirable style, such as being judgmental. Additionally, as model capabilities and usage patterns evolve, there may be a costly need to add or relabel data to modify safety behavior. We propose a novel preference modeling approach that utilizes AI feedback and","authors_text":"Alec Helyar, Alex Beutel, Andrea Vallone, Ian Kivlichan, Johannes Heidecke, John Schulman, Joshua Achiam, Lilian Weng, Molly Lin, Tong Mu","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-02T02:22:21Z","title":"Rule Based Rewards for Language Model Safety"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.01111","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:fd038064c3323f705dda3eb0d31d8bb926c7ac4a8266d0b86ed990184c15bf84","target":"record","created_at":"2026-07-05T09:30:24Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"05d6fc19b4dea9acdd4ebc65022ee98a927820abea0590b69560ca004eb879bc","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-02T02:22:21Z","title_canon_sha256":"712aa99b70285964f2d87066f20802020e775d1d71da8e75bbed74f37a41a0b0"},"schema_version":"1.0","source":{"id":"2411.01111","kind":"arxiv","version":1}},"canonical_sha256":"79e10f5b5ebbb1a1e82bf27f0b4a014af6b4a96aa66c6186f0d516fdeec7bf21","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"79e10f5b5ebbb1a1e82bf27f0b4a014af6b4a96aa66c6186f0d516fdeec7bf21","first_computed_at":"2026-07-05T09:30:24.868031Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:30:24.868031Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"yXzOoaobztNejStOg7zdzXT/OiDBJxnEVtwpWKJyH4wg92ImI88naPVUNNBv12MXFSkYkhpaNFBxxs2WJuCUBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T09:30:24.868545Z","signed_message":"canonical_sha256_bytes"},"source_id":"2411.01111","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:fd038064c3323f705dda3eb0d31d8bb926c7ac4a8266d0b86ed990184c15bf84","sha256:24bbddb3daff6ded2b60ee7c31c4df6083c73011145bcdc73dfd81846528e346"],"state_sha256":"99214878a070b34000ead804566b07e587311a9b4caa56bc9bead7265e073ba8"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JRIdjtYeWH5wZXi187JZ5sFkrp5mgQJHpaEPiJqpf+AIjjPtnQeE/QE6/jLPDW4XW0FieLDS774lMCSrB5B3CQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-01T11:02:30.109336Z","bundle_sha256":"c7bddfe701abf517d1eca8d80b0f0c26d403f53574c5fc6c971a088a63cb18c3"}}