{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TKJLJY72V6NKIXFWOUMZPVOV3B","short_pith_number":"pith:TKJLJY72","schema_version":"1.0","canonical_sha256":"9a92b4e3faaf9aa45cb6751997d5d5d853fd5865fa1c2658a180b6af00c96c54","source":{"kind":"arxiv","id":"2504.14177","version":1},"attestation_state":"computed","paper":{"title":"Direct Advantage Regression: Aligning LLMs with Online AI Reward","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.HC"],"primary_cat":"cs.AI","authors_text":"Dadong Wang, He Zhao, Li He, Lina Yao, Stephen Wan, Tongliang Liu","submitted_at":"2025-04-19T04:44:32Z","abstract_excerpt":"Online AI Feedback (OAIF) presents a promising alternative to Reinforcement Learning from Human Feedback (RLHF) by utilizing online AI preference in aligning language models (LLMs). However, the straightforward replacement of humans with AI deprives LLMs from learning more fine-grained AI supervision beyond binary signals. In this paper, we propose Direct Advantage Regression (DAR), a simple alignment algorithm using online AI reward to optimize policy improvement through weighted supervised fine-tuning. As an RL-free approach, DAR maintains theoretical consistency with online RLHF pipelines w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.14177","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-04-19T04:44:32Z","cross_cats_sorted":["cs.CL","cs.HC"],"title_canon_sha256":"96b9971ccb2f0cad30a2324545626c42f5e80eec292bfc22b834d0347e207007","abstract_canon_sha256":"a3e74eff4435b0ca81b7e6fae696a81ae6f5b8615cd44708de264a00b9ab0086"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:39.388906Z","signature_b64":"m2CRyCbhq7ZpWs02FBLdQBBWqzWZtk4k4p97V8Yk1B03vI/HXHf3nwqcyF4KstG5L9O2SbL7bxcusYP4xqZ8DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9a92b4e3faaf9aa45cb6751997d5d5d853fd5865fa1c2658a180b6af00c96c54","last_reissued_at":"2026-07-05T10:51:39.388417Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:39.388417Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Direct Advantage Regression: Aligning LLMs with Online AI Reward","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.HC"],"primary_cat":"cs.AI","authors_text":"Dadong Wang, He Zhao, Li He, Lina Yao, Stephen Wan, Tongliang Liu","submitted_at":"2025-04-19T04:44:32Z","abstract_excerpt":"Online AI Feedback (OAIF) presents a promising alternative to Reinforcement Learning from Human Feedback (RLHF) by utilizing online AI preference in aligning language models (LLMs). However, the straightforward replacement of humans with AI deprives LLMs from learning more fine-grained AI supervision beyond binary signals. In this paper, we propose Direct Advantage Regression (DAR), a simple alignment algorithm using online AI reward to optimize policy improvement through weighted supervised fine-tuning. As an RL-free approach, DAR maintains theoretical consistency with online RLHF pipelines w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14177","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.14177/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.14177","created_at":"2026-07-05T10:51:39.388486+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.14177v1","created_at":"2026-07-05T10:51:39.388486+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14177","created_at":"2026-07-05T10:51:39.388486+00:00"},{"alias_kind":"pith_short_12","alias_value":"TKJLJY72V6NK","created_at":"2026-07-05T10:51:39.388486+00:00"},{"alias_kind":"pith_short_16","alias_value":"TKJLJY72V6NKIXFW","created_at":"2026-07-05T10:51:39.388486+00:00"},{"alias_kind":"pith_short_8","alias_value":"TKJLJY72","created_at":"2026-07-05T10:51:39.388486+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TKJLJY72V6NKIXFWOUMZPVOV3B","json":"https://pith.science/pith/TKJLJY72V6NKIXFWOUMZPVOV3B.json","graph_json":"https://pith.science/api/pith-number/TKJLJY72V6NKIXFWOUMZPVOV3B/graph.json","events_json":"https://pith.science/api/pith-number/TKJLJY72V6NKIXFWOUMZPVOV3B/events.json","paper":"https://pith.science/paper/TKJLJY72"},"agent_actions":{"view_html":"https://pith.science/pith/TKJLJY72V6NKIXFWOUMZPVOV3B","download_json":"https://pith.science/pith/TKJLJY72V6NKIXFWOUMZPVOV3B.json","view_paper":"https://pith.science/paper/TKJLJY72","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.14177&json=true","fetch_graph":"https://pith.science/api/pith-number/TKJLJY72V6NKIXFWOUMZPVOV3B/graph.json","fetch_events":"https://pith.science/api/pith-number/TKJLJY72V6NKIXFWOUMZPVOV3B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TKJLJY72V6NKIXFWOUMZPVOV3B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TKJLJY72V6NKIXFWOUMZPVOV3B/action/storage_attestation","attest_author":"https://pith.science/pith/TKJLJY72V6NKIXFWOUMZPVOV3B/action/author_attestation","sign_citation":"https://pith.science/pith/TKJLJY72V6NKIXFWOUMZPVOV3B/action/citation_signature","submit_replication":"https://pith.science/pith/TKJLJY72V6NKIXFWOUMZPVOV3B/action/replication_record"}},"created_at":"2026-07-05T10:51:39.388486+00:00","updated_at":"2026-07-05T10:51:39.388486+00:00"}