{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:WFP4C4MWRSBDKGXKPXIR2SC7GB","short_pith_number":"pith:WFP4C4MW","canonical_record":{"source":{"id":"2401.10020","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-18T14:43:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ef0ddc650da67c1c350fe92f1ebb454b1f48bf675eb2cda43d4615e1807eda76","abstract_canon_sha256":"ceb6e5c3626454b73657ea05aa458d7b7fa15e3845ddd5b3b996f80ab57f296e"},"schema_version":"1.0"},"canonical_sha256":"b15fc171968c82351aea7dd11d485f3056deae3617143f7343904d9cd65cc4d8","source":{"kind":"arxiv","id":"2401.10020","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.10020","created_at":"2026-07-05T10:40:33Z"},{"alias_kind":"arxiv_version","alias_value":"2401.10020v3","created_at":"2026-07-05T10:40:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.10020","created_at":"2026-07-05T10:40:33Z"},{"alias_kind":"pith_short_12","alias_value":"WFP4C4MWRSBD","created_at":"2026-07-05T10:40:33Z"},{"alias_kind":"pith_short_16","alias_value":"WFP4C4MWRSBDKGXK","created_at":"2026-07-05T10:40:33Z"},{"alias_kind":"pith_short_8","alias_value":"WFP4C4MW","created_at":"2026-07-05T10:40:33Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:WFP4C4MWRSBDKGXKPXIR2SC7GB","target":"record","payload":{"canonical_record":{"source":{"id":"2401.10020","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-18T14:43:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ef0ddc650da67c1c350fe92f1ebb454b1f48bf675eb2cda43d4615e1807eda76","abstract_canon_sha256":"ceb6e5c3626454b73657ea05aa458d7b7fa15e3845ddd5b3b996f80ab57f296e"},"schema_version":"1.0"},"canonical_sha256":"b15fc171968c82351aea7dd11d485f3056deae3617143f7343904d9cd65cc4d8","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:33.028737Z","signature_b64":"iCmyZ1Vb3os1Hn8DYlywUMgHZZ5GKQiK7ElaH5h/AcNNnzyQg94BwUmUSN66Zbua5la2/QkYcEPoblFJQcX7AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b15fc171968c82351aea7dd11d485f3056deae3617143f7343904d9cd65cc4d8","last_reissued_at":"2026-07-05T10:40:33.028291Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:33.028291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2401.10020","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:40:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"YVnXRnTJ1lNVO4gtzRSWEMaBiiMFrhy3X5vLHlIIZCGZuZtlnApZNiwS1LscvkNzIA9JHXIlXb7t7C8qwc0+Aw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T23:15:26.771414Z"},"content_sha256":"f28a8e9ae10855e75f8a953209e6740a6f932dffde81033a61378f415bbec6e2","schema_version":"1.0","event_id":"sha256:f28a8e9ae10855e75f8a953209e6740a6f932dffde81033a61378f415bbec6e2"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:WFP4C4MWRSBDKGXKPXIR2SC7GB","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Self-Rewarding Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Language models can train themselves by using their own judgments to generate rewards for iterative improvement.","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jason Weston, Jing Xu, Kyunghyun Cho, Richard Yuanzhe Pang, Sainbayar Sukhbaatar, Weizhe Yuan, Xian Li","submitted_at":"2024-01-18T14:43:47Z","abstract_excerpt":"We posit that to achieve superhuman agents, future models require superhuman feedback in order to provide an adequate training signal. Current approaches commonly train reward models from human preferences, which may then be bottlenecked by human performance level, and secondly these separate frozen reward models cannot then learn to improve during LLM training. In this work, we study Self-Rewarding Language Models, where the language model itself is used via LLM-as-a-Judge prompting to provide its own rewards during training. We show that during Iterative DPO training that not only does instr"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Fine-tuning Llama 2 70B on three iterations of our approach yields a model that outperforms many existing systems on the AlpacaEval 2.0 leaderboard, including Claude 2, Gemini Pro, and GPT-4 0613.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The LLM-as-a-Judge prompting produces reliable, unbiased rewards that drive genuine capability gains rather than self-reinforcing errors or biases in the model's own judgments.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Iterative self-rewarding via LLM-as-Judge in DPO training on Llama 2 70B improves instruction following and self-evaluation, outperforming GPT-4 on AlpacaEval 2.0.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Language models can train themselves by using their own judgments to generate rewards for iterative improvement.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"1c759b75486b5defaadcc3d61267de2312013537a6c7048f730faefa41379a04"},"source":{"id":"2401.10020","kind":"arxiv","version":3},"verdict":{"id":"8b97fc9f-109f-495e-8903-cbfe619fcf0c","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-13T11:57:27.063100Z","strongest_claim":"Fine-tuning Llama 2 70B on three iterations of our approach yields a model that outperforms many existing systems on the AlpacaEval 2.0 leaderboard, including Claude 2, Gemini Pro, and GPT-4 0613.","one_line_summary":"Iterative self-rewarding via LLM-as-Judge in DPO training on Llama 2 70B improves instruction following and self-evaluation, outperforming GPT-4 on AlpacaEval 2.0.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The LLM-as-a-Judge prompting produces reliable, unbiased rewards that drive genuine capability gains rather than self-reinforcing errors or biases in the model's own judgments.","pith_extraction_headline":"Language models can train themselves by using their own judgments to generate rewards for iterative improvement."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.10020/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":104,"sample":[{"doi":"","year":null,"title":"Advances in Neural Information Processing Systems , volume=","work_id":"bd577a47-49ef-4f8a-81ca-86cd88b71479","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"Think you have solved question answering?","work_id":"65ee9a0a-7b80-4a03-a16e-5467747a1e24","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2019,"title":"2019 , journal =","work_id":"41c879fa-f0bb-4f50-b5be-c2b0d1b21ffa","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"Can a Suit of Armor Conduct Electricity? A New Dataset for Open Book Question Answering , author=. EMNLP , year=","work_id":"5928ce4f-4b10-4358-9287-a8ba1af28098","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2021,"title":"9th International Conference on Learning Representations","work_id":"71e363a8-ff38-4e5f-b603-9e3c1f32e4a0","ref_index":6,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":104,"snapshot_sha256":"d589f2bc43bc383294799ae5b431984f6f09f661fb0ddf727bbbe455f0416a6f","internal_anchors":29},"formal_canon":{"evidence_count":2,"snapshot_sha256":"025ec951f10cbeba5206f7daf6616bb22519173182eac9d481ebbe8c9eb521f1"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"8b97fc9f-109f-495e-8903-cbfe619fcf0c"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:40:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jxtbO5Vqm2vde4VZs7SDiLhOCyXIcZA+V3yHjFzkC31MPxEdZmekkKwQ45d8xTPsz8kbNsYehVQUZVPVyRH8BA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T23:15:26.772245Z"},"content_sha256":"39a406fd8eb50b8f41f89030c730fd76df552d099adc11e1a5e997c0c9f0b400","schema_version":"1.0","event_id":"sha256:39a406fd8eb50b8f41f89030c730fd76df552d099adc11e1a5e997c0c9f0b400"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/WFP4C4MWRSBDKGXKPXIR2SC7GB/bundle.json","state_url":"https://pith.science/pith/WFP4C4MWRSBDKGXKPXIR2SC7GB/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/WFP4C4MWRSBDKGXKPXIR2SC7GB/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T23:15:26Z","links":{"resolver":"https://pith.science/pith/WFP4C4MWRSBDKGXKPXIR2SC7GB","bundle":"https://pith.science/pith/WFP4C4MWRSBDKGXKPXIR2SC7GB/bundle.json","state":"https://pith.science/pith/WFP4C4MWRSBDKGXKPXIR2SC7GB/state.json","well_known_bundle":"https://pith.science/.well-known/pith/WFP4C4MWRSBDKGXKPXIR2SC7GB/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:WFP4C4MWRSBDKGXKPXIR2SC7GB","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ceb6e5c3626454b73657ea05aa458d7b7fa15e3845ddd5b3b996f80ab57f296e","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-18T14:43:47Z","title_canon_sha256":"ef0ddc650da67c1c350fe92f1ebb454b1f48bf675eb2cda43d4615e1807eda76"},"schema_version":"1.0","source":{"id":"2401.10020","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.10020","created_at":"2026-07-05T10:40:33Z"},{"alias_kind":"arxiv_version","alias_value":"2401.10020v3","created_at":"2026-07-05T10:40:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.10020","created_at":"2026-07-05T10:40:33Z"},{"alias_kind":"pith_short_12","alias_value":"WFP4C4MWRSBD","created_at":"2026-07-05T10:40:33Z"},{"alias_kind":"pith_short_16","alias_value":"WFP4C4MWRSBDKGXK","created_at":"2026-07-05T10:40:33Z"},{"alias_kind":"pith_short_8","alias_value":"WFP4C4MW","created_at":"2026-07-05T10:40:33Z"}],"graph_snapshots":[{"event_id":"sha256:39a406fd8eb50b8f41f89030c730fd76df552d099adc11e1a5e997c0c9f0b400","target":"graph","created_at":"2026-07-05T10:40:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Fine-tuning Llama 2 70B on three iterations of our approach yields a model that outperforms many existing systems on the AlpacaEval 2.0 leaderboard, including Claude 2, Gemini Pro, and GPT-4 0613."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The LLM-as-a-Judge prompting produces reliable, unbiased rewards that drive genuine capability gains rather than self-reinforcing errors or biases in the model's own judgments."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Iterative self-rewarding via LLM-as-Judge in DPO training on Llama 2 70B improves instruction following and self-evaluation, outperforming GPT-4 on AlpacaEval 2.0."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Language models can train themselves by using their own judgments to generate rewards for iterative improvement."}],"snapshot_sha256":"1c759b75486b5defaadcc3d61267de2312013537a6c7048f730faefa41379a04"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"025ec951f10cbeba5206f7daf6616bb22519173182eac9d481ebbe8c9eb521f1"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2401.10020/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We posit that to achieve superhuman agents, future models require superhuman feedback in order to provide an adequate training signal. Current approaches commonly train reward models from human preferences, which may then be bottlenecked by human performance level, and secondly these separate frozen reward models cannot then learn to improve during LLM training. In this work, we study Self-Rewarding Language Models, where the language model itself is used via LLM-as-a-Judge prompting to provide its own rewards during training. We show that during Iterative DPO training that not only does instr","authors_text":"Jason Weston, Jing Xu, Kyunghyun Cho, Richard Yuanzhe Pang, Sainbayar Sukhbaatar, Weizhe Yuan, Xian Li","cross_cats":["cs.AI"],"headline":"Language models can train themselves by using their own judgments to generate rewards for iterative improvement.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-18T14:43:47Z","title":"Self-Rewarding Language Models"},"references":{"count":104,"internal_anchors":29,"resolved_work":104,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"Advances in Neural Information Processing Systems , volume=","work_id":"bd577a47-49ef-4f8a-81ca-86cd88b71479","year":null},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"Think you have solved question answering?","work_id":"65ee9a0a-7b80-4a03-a16e-5467747a1e24","year":null},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":4,"title":"2019 , journal =","work_id":"41c879fa-f0bb-4f50-b5be-c2b0d1b21ffa","year":2019},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Can a Suit of Armor Conduct Electricity? A New Dataset for Open Book Question Answering , author=. EMNLP , year=","work_id":"5928ce4f-4b10-4358-9287-a8ba1af28098","year":null},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":6,"title":"9th International Conference on Learning Representations","work_id":"71e363a8-ff38-4e5f-b603-9e3c1f32e4a0","year":2021}],"snapshot_sha256":"d589f2bc43bc383294799ae5b431984f6f09f661fb0ddf727bbbe455f0416a6f"},"source":{"id":"2401.10020","kind":"arxiv","version":3},"verdict":{"created_at":"2026-05-13T11:57:27.063100Z","id":"8b97fc9f-109f-495e-8903-cbfe619fcf0c","model_set":{"reader":"grok-4.3"},"one_line_summary":"Iterative self-rewarding via LLM-as-Judge in DPO training on Llama 2 70B improves instruction following and self-evaluation, outperforming GPT-4 on AlpacaEval 2.0.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Language models can train themselves by using their own judgments to generate rewards for iterative improvement.","strongest_claim":"Fine-tuning Llama 2 70B on three iterations of our approach yields a model that outperforms many existing systems on the AlpacaEval 2.0 leaderboard, including Claude 2, Gemini Pro, and GPT-4 0613.","weakest_assumption":"The LLM-as-a-Judge prompting produces reliable, unbiased rewards that drive genuine capability gains rather than self-reinforcing errors or biases in the model's own judgments."}},"verdict_id":"8b97fc9f-109f-495e-8903-cbfe619fcf0c"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f28a8e9ae10855e75f8a953209e6740a6f932dffde81033a61378f415bbec6e2","target":"record","created_at":"2026-07-05T10:40:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ceb6e5c3626454b73657ea05aa458d7b7fa15e3845ddd5b3b996f80ab57f296e","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-18T14:43:47Z","title_canon_sha256":"ef0ddc650da67c1c350fe92f1ebb454b1f48bf675eb2cda43d4615e1807eda76"},"schema_version":"1.0","source":{"id":"2401.10020","kind":"arxiv","version":3}},"canonical_sha256":"b15fc171968c82351aea7dd11d485f3056deae3617143f7343904d9cd65cc4d8","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b15fc171968c82351aea7dd11d485f3056deae3617143f7343904d9cd65cc4d8","first_computed_at":"2026-07-05T10:40:33.028291Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:40:33.028291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"iCmyZ1Vb3os1Hn8DYlywUMgHZZ5GKQiK7ElaH5h/AcNNnzyQg94BwUmUSN66Zbua5la2/QkYcEPoblFJQcX7AQ==","signature_status":"signed_v1","signed_at":"2026-07-05T10:40:33.028737Z","signed_message":"canonical_sha256_bytes"},"source_id":"2401.10020","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f28a8e9ae10855e75f8a953209e6740a6f932dffde81033a61378f415bbec6e2","sha256:39a406fd8eb50b8f41f89030c730fd76df552d099adc11e1a5e997c0c9f0b400"],"state_sha256":"752b264f5c8a41795fcc730f0a7a3eb383dc7b1e85e49566567bce65841e981b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"sddoYY+XVCMoqjoY7LEPTx2oRJSsTHxffsgY2ygaOBHCS1ZpS4Se/dBfcz4lbZbc36F1s3YpKOivfVdreqYWBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T23:15:26.776850Z","bundle_sha256":"5f2fd959424efe12851685043b05f34909e9da76b8459148b2dfacbd0619858d"}}