{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:TJ7G7O7H7LMAQVQUZ2DS7W5YLT","short_pith_number":"pith:TJ7G7O7H","canonical_record":{"source":{"id":"2306.02231","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-04T01:59:40Z","cross_cats_sorted":["cs.AI","cs.LG","cs.SY","eess.SY"],"title_canon_sha256":"39923e9a8e3b60d94aea3ceba26b31bcf489b283f5db2f3d2a0c24c239b220f4","abstract_canon_sha256":"9f4291b82ba4cefd32cb62da4f94894317be1d7a6384a7ac2fb1f952e33bb6d4"},"schema_version":"1.0"},"canonical_sha256":"9a7e6fbbe7fad8085614ce872fdbb85cc0bb93639f5360d9cd1cf3ed508c0093","source":{"kind":"arxiv","id":"2306.02231","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2306.02231","created_at":"2026-07-05T07:08:28Z"},{"alias_kind":"arxiv_version","alias_value":"2306.02231v3","created_at":"2026-07-05T07:08:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.02231","created_at":"2026-07-05T07:08:28Z"},{"alias_kind":"pith_short_12","alias_value":"TJ7G7O7H7LMA","created_at":"2026-07-05T07:08:28Z"},{"alias_kind":"pith_short_16","alias_value":"TJ7G7O7H7LMAQVQU","created_at":"2026-07-05T07:08:28Z"},{"alias_kind":"pith_short_8","alias_value":"TJ7G7O7H","created_at":"2026-07-05T07:08:28Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:TJ7G7O7H7LMAQVQUZ2DS7W5YLT","target":"record","payload":{"canonical_record":{"source":{"id":"2306.02231","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-04T01:59:40Z","cross_cats_sorted":["cs.AI","cs.LG","cs.SY","eess.SY"],"title_canon_sha256":"39923e9a8e3b60d94aea3ceba26b31bcf489b283f5db2f3d2a0c24c239b220f4","abstract_canon_sha256":"9f4291b82ba4cefd32cb62da4f94894317be1d7a6384a7ac2fb1f952e33bb6d4"},"schema_version":"1.0"},"canonical_sha256":"9a7e6fbbe7fad8085614ce872fdbb85cc0bb93639f5360d9cd1cf3ed508c0093","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:08:28.709678Z","signature_b64":"sVAYiROd9MEiKfVl5TFXMdu01OHdR/VuF9NydkXs7C9mxqLqd1mU6+6zOYG2gfkDGM3UfMFxGz1B2tATi7VUBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9a7e6fbbe7fad8085614ce872fdbb85cc0bb93639f5360d9cd1cf3ed508c0093","last_reissued_at":"2026-07-05T07:08:28.709170Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:08:28.709170Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2306.02231","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:08:28Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"NFPOpqUu2m9DNjgFQwRqLzKw1DymYsJ30i8ta2AduhNeoH0Ej0N6cmPfGoOxz/1xcF5UrMxtjp0gJJ6EJ6QfCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T03:44:00.529406Z"},"content_sha256":"11084349fc80ae850ac3e041ed03f6c9ba1cc1a240085fa696869737352098f6","schema_version":"1.0","event_id":"sha256:11084349fc80ae850ac3e041ed03f6c9ba1cc1a240085fa696869737352098f6"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:TJ7G7O7H7LMAQVQUZ2DS7W5YLT","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Fine-Tuning Language Models with Advantage-Induced Policy Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SY","eess.SY"],"primary_cat":"cs.CL","authors_text":"Banghua Zhu, Chenguang Zhu, Felipe Vieira Frujeri, Hiteshi Sharma, Jiantao Jiao, Michael I. Jordan, Shi Dong","submitted_at":"2023-06-04T01:59:40Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has emerged as a reliable approach to aligning large language models (LLMs) to human preferences. Among the plethora of RLHF techniques, proximal policy optimization (PPO) is of the most widely used methods. Despite its popularity, however, PPO may suffer from mode collapse, instability, and poor sample efficiency. We show that these issues can be alleviated by a novel algorithm that we refer to as Advantage-Induced Policy Alignment (APA), which leverages a squared error loss function based on the estimated advantages. We demonstrate empiricall"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.02231","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.02231/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:08:28Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CS54O5R4twogD7cbkBCx9fqxQ2nscTCCuHLzomi5vSMC+4aLQD4zoBDEJkTxF708Fldp3K9WZFQ1FB7Z49LeCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T03:44:00.529957Z"},"content_sha256":"54d25b2021c167166f3d2f5b6c92133d46f9e65b7c1c565d401a4d01ac9a786e","schema_version":"1.0","event_id":"sha256:54d25b2021c167166f3d2f5b6c92133d46f9e65b7c1c565d401a4d01ac9a786e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/TJ7G7O7H7LMAQVQUZ2DS7W5YLT/bundle.json","state_url":"https://pith.science/pith/TJ7G7O7H7LMAQVQUZ2DS7W5YLT/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/TJ7G7O7H7LMAQVQUZ2DS7W5YLT/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T03:44:00Z","links":{"resolver":"https://pith.science/pith/TJ7G7O7H7LMAQVQUZ2DS7W5YLT","bundle":"https://pith.science/pith/TJ7G7O7H7LMAQVQUZ2DS7W5YLT/bundle.json","state":"https://pith.science/pith/TJ7G7O7H7LMAQVQUZ2DS7W5YLT/state.json","well_known_bundle":"https://pith.science/.well-known/pith/TJ7G7O7H7LMAQVQUZ2DS7W5YLT/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:TJ7G7O7H7LMAQVQUZ2DS7W5YLT","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"9f4291b82ba4cefd32cb62da4f94894317be1d7a6384a7ac2fb1f952e33bb6d4","cross_cats_sorted":["cs.AI","cs.LG","cs.SY","eess.SY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-04T01:59:40Z","title_canon_sha256":"39923e9a8e3b60d94aea3ceba26b31bcf489b283f5db2f3d2a0c24c239b220f4"},"schema_version":"1.0","source":{"id":"2306.02231","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2306.02231","created_at":"2026-07-05T07:08:28Z"},{"alias_kind":"arxiv_version","alias_value":"2306.02231v3","created_at":"2026-07-05T07:08:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.02231","created_at":"2026-07-05T07:08:28Z"},{"alias_kind":"pith_short_12","alias_value":"TJ7G7O7H7LMA","created_at":"2026-07-05T07:08:28Z"},{"alias_kind":"pith_short_16","alias_value":"TJ7G7O7H7LMAQVQU","created_at":"2026-07-05T07:08:28Z"},{"alias_kind":"pith_short_8","alias_value":"TJ7G7O7H","created_at":"2026-07-05T07:08:28Z"}],"graph_snapshots":[{"event_id":"sha256:54d25b2021c167166f3d2f5b6c92133d46f9e65b7c1c565d401a4d01ac9a786e","target":"graph","created_at":"2026-07-05T07:08:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2306.02231/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has emerged as a reliable approach to aligning large language models (LLMs) to human preferences. Among the plethora of RLHF techniques, proximal policy optimization (PPO) is of the most widely used methods. Despite its popularity, however, PPO may suffer from mode collapse, instability, and poor sample efficiency. We show that these issues can be alleviated by a novel algorithm that we refer to as Advantage-Induced Policy Alignment (APA), which leverages a squared error loss function based on the estimated advantages. We demonstrate empiricall","authors_text":"Banghua Zhu, Chenguang Zhu, Felipe Vieira Frujeri, Hiteshi Sharma, Jiantao Jiao, Michael I. Jordan, Shi Dong","cross_cats":["cs.AI","cs.LG","cs.SY","eess.SY"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-04T01:59:40Z","title":"Fine-Tuning Language Models with Advantage-Induced Policy Alignment"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.02231","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:11084349fc80ae850ac3e041ed03f6c9ba1cc1a240085fa696869737352098f6","target":"record","created_at":"2026-07-05T07:08:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"9f4291b82ba4cefd32cb62da4f94894317be1d7a6384a7ac2fb1f952e33bb6d4","cross_cats_sorted":["cs.AI","cs.LG","cs.SY","eess.SY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-04T01:59:40Z","title_canon_sha256":"39923e9a8e3b60d94aea3ceba26b31bcf489b283f5db2f3d2a0c24c239b220f4"},"schema_version":"1.0","source":{"id":"2306.02231","kind":"arxiv","version":3}},"canonical_sha256":"9a7e6fbbe7fad8085614ce872fdbb85cc0bb93639f5360d9cd1cf3ed508c0093","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9a7e6fbbe7fad8085614ce872fdbb85cc0bb93639f5360d9cd1cf3ed508c0093","first_computed_at":"2026-07-05T07:08:28.709170Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:08:28.709170Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"sVAYiROd9MEiKfVl5TFXMdu01OHdR/VuF9NydkXs7C9mxqLqd1mU6+6zOYG2gfkDGM3UfMFxGz1B2tATi7VUBw==","signature_status":"signed_v1","signed_at":"2026-07-05T07:08:28.709678Z","signed_message":"canonical_sha256_bytes"},"source_id":"2306.02231","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:11084349fc80ae850ac3e041ed03f6c9ba1cc1a240085fa696869737352098f6","sha256:54d25b2021c167166f3d2f5b6c92133d46f9e65b7c1c565d401a4d01ac9a786e"],"state_sha256":"0a109fb8817b99f1e5f07ceae023e2b752bf47ddac439d9de0b6d6f56c6bb418"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"6gZG0HfxKZuMlwb8RmvWXJEDwkCkUJIZBhfUbitUIR/7w5umZA/iqZdj7KCbyPjRIHsCKwSS3KnIcaoTINDjBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T03:44:00.535671Z","bundle_sha256":"cd825be14454d2903cc121b5d2ac296d7df1acdb3f2216e2ffa67729dbf49c97"}}