{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:4PJHHX5JZAJDWUAKV6XVVTCAHF","short_pith_number":"pith:4PJHHX5J","canonical_record":{"source":{"id":"2404.08555","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-12T15:54:15Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"149bdaa31c2cea2828bc6e837a7e1ffc686360878bea9143e1eb9b74bf520a61","abstract_canon_sha256":"608cd5a355fd62edc5cdf67bf9ad900de2328272f9f757fffa67ff177d96667e"},"schema_version":"1.0"},"canonical_sha256":"e3d273dfa9c8123b500aafaf5acc40396acc1097e34885e76db196d0c3bfa82f","source":{"kind":"arxiv","id":"2404.08555","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.08555","created_at":"2026-07-05T08:08:23Z"},{"alias_kind":"arxiv_version","alias_value":"2404.08555v2","created_at":"2026-07-05T08:08:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.08555","created_at":"2026-07-05T08:08:23Z"},{"alias_kind":"pith_short_12","alias_value":"4PJHHX5JZAJD","created_at":"2026-07-05T08:08:23Z"},{"alias_kind":"pith_short_16","alias_value":"4PJHHX5JZAJDWUAK","created_at":"2026-07-05T08:08:23Z"},{"alias_kind":"pith_short_8","alias_value":"4PJHHX5J","created_at":"2026-07-05T08:08:23Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:4PJHHX5JZAJDWUAKV6XVVTCAHF","target":"record","payload":{"canonical_record":{"source":{"id":"2404.08555","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-12T15:54:15Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"149bdaa31c2cea2828bc6e837a7e1ffc686360878bea9143e1eb9b74bf520a61","abstract_canon_sha256":"608cd5a355fd62edc5cdf67bf9ad900de2328272f9f757fffa67ff177d96667e"},"schema_version":"1.0"},"canonical_sha256":"e3d273dfa9c8123b500aafaf5acc40396acc1097e34885e76db196d0c3bfa82f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:23.256930Z","signature_b64":"5hsoonRDZn41L2uv/8l/oatwEy6RD+cuAkM+fhz1z63Z+5Mr8xoe2t5u0tCT7t6J7ae7wwzp8r0od/4XKqX5Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e3d273dfa9c8123b500aafaf5acc40396acc1097e34885e76db196d0c3bfa82f","last_reissued_at":"2026-07-05T08:08:23.256458Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:23.256458Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2404.08555","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:08:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/SUdTrtft3WM2Ot0QkS1vyZOvj0PEXRMYTfr6yLrh+uKeSp5RAn/x/YUxzsVtFyXiwzrqa6ObmAHMJn1LCZ1BQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-14T10:35:00.615943Z"},"content_sha256":"8ad6e2736105d61bb12901acbdc6cdb887f76940ff145be33b336a3e74524fd5","schema_version":"1.0","event_id":"sha256:8ad6e2736105d61bb12901acbdc6cdb887f76940ff145be33b336a3e74524fd5"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:4PJHHX5JZAJDWUAKV6XVVTCAHF","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"RLHF Deciphered: A Critical Analysis of Reinforcement Learning from Human Feedback for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Ameet Deshpande, Ashwin Kalyan, Bruno Castro da Silva, Karthik Narasimhan, Pranjal Aggarwal, Shreyas Chaudhari, Tanmay Rajpurohit, Vishvak Murahari","submitted_at":"2024-04-12T15:54:15Z","abstract_excerpt":"State-of-the-art large language models (LLMs) have become indispensable tools for various tasks. However, training LLMs to serve as effective assistants for humans requires careful consideration. A promising approach is reinforcement learning from human feedback (RLHF), which leverages human feedback to update the model in accordance with human preferences and mitigate issues like toxicity and hallucinations. Yet, an understanding of RLHF for LLMs is largely entangled with initial design choices that popularized the method and current research focuses on augmenting those choices rather than fu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.08555","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.08555/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:08:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"uQtpjG7ipqII7BliJXXoZHNTB4MnFHQnmUtsPXlZoKCsKsskK5fUn5aRWStQB0GesRH6eC0aUQGRUOOx7LQ0BA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-14T10:35:00.616459Z"},"content_sha256":"1ce14278a850b4ddc076343f7f99d8ae260b0e025282093c2f71cdd44ae1be31","schema_version":"1.0","event_id":"sha256:1ce14278a850b4ddc076343f7f99d8ae260b0e025282093c2f71cdd44ae1be31"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/4PJHHX5JZAJDWUAKV6XVVTCAHF/bundle.json","state_url":"https://pith.science/pith/4PJHHX5JZAJDWUAKV6XVVTCAHF/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/4PJHHX5JZAJDWUAKV6XVVTCAHF/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-14T10:35:00Z","links":{"resolver":"https://pith.science/pith/4PJHHX5JZAJDWUAKV6XVVTCAHF","bundle":"https://pith.science/pith/4PJHHX5JZAJDWUAKV6XVVTCAHF/bundle.json","state":"https://pith.science/pith/4PJHHX5JZAJDWUAKV6XVVTCAHF/state.json","well_known_bundle":"https://pith.science/.well-known/pith/4PJHHX5JZAJDWUAKV6XVVTCAHF/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:4PJHHX5JZAJDWUAKV6XVVTCAHF","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"608cd5a355fd62edc5cdf67bf9ad900de2328272f9f757fffa67ff177d96667e","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-12T15:54:15Z","title_canon_sha256":"149bdaa31c2cea2828bc6e837a7e1ffc686360878bea9143e1eb9b74bf520a61"},"schema_version":"1.0","source":{"id":"2404.08555","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.08555","created_at":"2026-07-05T08:08:23Z"},{"alias_kind":"arxiv_version","alias_value":"2404.08555v2","created_at":"2026-07-05T08:08:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.08555","created_at":"2026-07-05T08:08:23Z"},{"alias_kind":"pith_short_12","alias_value":"4PJHHX5JZAJD","created_at":"2026-07-05T08:08:23Z"},{"alias_kind":"pith_short_16","alias_value":"4PJHHX5JZAJDWUAK","created_at":"2026-07-05T08:08:23Z"},{"alias_kind":"pith_short_8","alias_value":"4PJHHX5J","created_at":"2026-07-05T08:08:23Z"}],"graph_snapshots":[{"event_id":"sha256:1ce14278a850b4ddc076343f7f99d8ae260b0e025282093c2f71cdd44ae1be31","target":"graph","created_at":"2026-07-05T08:08:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2404.08555/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"State-of-the-art large language models (LLMs) have become indispensable tools for various tasks. However, training LLMs to serve as effective assistants for humans requires careful consideration. A promising approach is reinforcement learning from human feedback (RLHF), which leverages human feedback to update the model in accordance with human preferences and mitigate issues like toxicity and hallucinations. Yet, an understanding of RLHF for LLMs is largely entangled with initial design choices that popularized the method and current research focuses on augmenting those choices rather than fu","authors_text":"Ameet Deshpande, Ashwin Kalyan, Bruno Castro da Silva, Karthik Narasimhan, Pranjal Aggarwal, Shreyas Chaudhari, Tanmay Rajpurohit, Vishvak Murahari","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-12T15:54:15Z","title":"RLHF Deciphered: A Critical Analysis of Reinforcement Learning from Human Feedback for LLMs"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.08555","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:8ad6e2736105d61bb12901acbdc6cdb887f76940ff145be33b336a3e74524fd5","target":"record","created_at":"2026-07-05T08:08:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"608cd5a355fd62edc5cdf67bf9ad900de2328272f9f757fffa67ff177d96667e","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-12T15:54:15Z","title_canon_sha256":"149bdaa31c2cea2828bc6e837a7e1ffc686360878bea9143e1eb9b74bf520a61"},"schema_version":"1.0","source":{"id":"2404.08555","kind":"arxiv","version":2}},"canonical_sha256":"e3d273dfa9c8123b500aafaf5acc40396acc1097e34885e76db196d0c3bfa82f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e3d273dfa9c8123b500aafaf5acc40396acc1097e34885e76db196d0c3bfa82f","first_computed_at":"2026-07-05T08:08:23.256458Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:08:23.256458Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"5hsoonRDZn41L2uv/8l/oatwEy6RD+cuAkM+fhz1z63Z+5Mr8xoe2t5u0tCT7t6J7ae7wwzp8r0od/4XKqX5Ag==","signature_status":"signed_v1","signed_at":"2026-07-05T08:08:23.256930Z","signed_message":"canonical_sha256_bytes"},"source_id":"2404.08555","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:8ad6e2736105d61bb12901acbdc6cdb887f76940ff145be33b336a3e74524fd5","sha256:1ce14278a850b4ddc076343f7f99d8ae260b0e025282093c2f71cdd44ae1be31"],"state_sha256":"673d74607d34a6009fe6d646bc75443bfbd1c5f2cc4c4cf2dbeaa3edb94f69d0"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VyDVfLjct1+2+QamUCYhoLPOprIXZWDyrXK94bW3oBpnoY10QzQcOd6tkcTGgWKo3Kc7wY/XKuwk8afbUfKXAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-14T10:35:00.620681Z","bundle_sha256":"bf75ee441e48dc86adffbcf9d0b5545afc17f2377d93f3860c90d9048933acb6"}}