{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NL6BJQNDF5BHIAJASQ7IKJTYBP","short_pith_number":"pith:NL6BJQND","schema_version":"1.0","canonical_sha256":"6afc14c1a32f42740120943e8526780bdb04cf5657735b64c260e808e4abf002","source":{"kind":"arxiv","id":"2305.06176","version":3},"attestation_state":"computed","paper":{"title":"Fine-tuning Language Models with Generative Adversarial Reward Modelling","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bryan Kian Hsiang Low, Lau Jia Jaw, Zhang Hui, Zhang Ze Yu","submitted_at":"2023-05-09T17:06:06Z","abstract_excerpt":"Reinforcement Learning with Human Feedback (RLHF) has been demonstrated to significantly enhance the performance of large language models (LLMs) by aligning their outputs with desired human values through instruction tuning. However, RLHF is constrained by the expertise and productivity limitations of human evaluators. A response to this downside is to fall back to supervised fine-tuning (SFT) with additional carefully selected expert demonstrations. However, while this method has been proven to be effective, it invariably also leads to increased human-in-the-loop overhead. In this study, we p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.06176","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-09T17:06:06Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8d21c5fd40bd65dd0aa632e2056da864cbb397ec18b36407c9e3a147f7f5dd00","abstract_canon_sha256":"09c11cd8cf548fd9cabed144f0ea0048f8828c9be8cab56ecfb99eac76376323"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:52:09.676834Z","signature_b64":"2sswQgRcLf7DHklB9jx0tWkazFYoVhaAyPgUGnlMbM3ciVUHdF0JDDfbsyQcCTqhtAs04OvuTT6f+2fQeBAEBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6afc14c1a32f42740120943e8526780bdb04cf5657735b64c260e808e4abf002","last_reissued_at":"2026-07-05T07:52:09.676388Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:52:09.676388Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fine-tuning Language Models with Generative Adversarial Reward Modelling","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bryan Kian Hsiang Low, Lau Jia Jaw, Zhang Hui, Zhang Ze Yu","submitted_at":"2023-05-09T17:06:06Z","abstract_excerpt":"Reinforcement Learning with Human Feedback (RLHF) has been demonstrated to significantly enhance the performance of large language models (LLMs) by aligning their outputs with desired human values through instruction tuning. However, RLHF is constrained by the expertise and productivity limitations of human evaluators. A response to this downside is to fall back to supervised fine-tuning (SFT) with additional carefully selected expert demonstrations. However, while this method has been proven to be effective, it invariably also leads to increased human-in-the-loop overhead. In this study, we p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.06176","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.06176/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.06176","created_at":"2026-07-05T07:52:09.676455+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.06176v3","created_at":"2026-07-05T07:52:09.676455+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.06176","created_at":"2026-07-05T07:52:09.676455+00:00"},{"alias_kind":"pith_short_12","alias_value":"NL6BJQNDF5BH","created_at":"2026-07-05T07:52:09.676455+00:00"},{"alias_kind":"pith_short_16","alias_value":"NL6BJQNDF5BHIAJA","created_at":"2026-07-05T07:52:09.676455+00:00"},{"alias_kind":"pith_short_8","alias_value":"NL6BJQND","created_at":"2026-07-05T07:52:09.676455+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.17879","citing_title":"Generative Adversarial Post-Training Mitigates Reward Hacking in Live Human-AI Music Interaction","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NL6BJQNDF5BHIAJASQ7IKJTYBP","json":"https://pith.science/pith/NL6BJQNDF5BHIAJASQ7IKJTYBP.json","graph_json":"https://pith.science/api/pith-number/NL6BJQNDF5BHIAJASQ7IKJTYBP/graph.json","events_json":"https://pith.science/api/pith-number/NL6BJQNDF5BHIAJASQ7IKJTYBP/events.json","paper":"https://pith.science/paper/NL6BJQND"},"agent_actions":{"view_html":"https://pith.science/pith/NL6BJQNDF5BHIAJASQ7IKJTYBP","download_json":"https://pith.science/pith/NL6BJQNDF5BHIAJASQ7IKJTYBP.json","view_paper":"https://pith.science/paper/NL6BJQND","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.06176&json=true","fetch_graph":"https://pith.science/api/pith-number/NL6BJQNDF5BHIAJASQ7IKJTYBP/graph.json","fetch_events":"https://pith.science/api/pith-number/NL6BJQNDF5BHIAJASQ7IKJTYBP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NL6BJQNDF5BHIAJASQ7IKJTYBP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NL6BJQNDF5BHIAJASQ7IKJTYBP/action/storage_attestation","attest_author":"https://pith.science/pith/NL6BJQNDF5BHIAJASQ7IKJTYBP/action/author_attestation","sign_citation":"https://pith.science/pith/NL6BJQNDF5BHIAJASQ7IKJTYBP/action/citation_signature","submit_replication":"https://pith.science/pith/NL6BJQNDF5BHIAJASQ7IKJTYBP/action/replication_record"}},"created_at":"2026-07-05T07:52:09.676455+00:00","updated_at":"2026-07-05T07:52:09.676455+00:00"}