{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BGFA4ICRWHNGMAKEP72GIFR72D","short_pith_number":"pith:BGFA4ICR","schema_version":"1.0","canonical_sha256":"098a0e2051b1da6601447ff464163fd0e66114336f87e9e455fea53803ef473d","source":{"kind":"arxiv","id":"2303.16755","version":3},"attestation_state":"computed","paper":{"title":"Training Language Models with Language Feedback at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Angelica Chen, Ethan Perez, J\\'er\\'emy Scheurer, Jon Ander Campos, Jun Shern Chan, Kyunghyun Cho, Tomasz Korbak","submitted_at":"2023-03-28T17:04:15Z","abstract_excerpt":"Pretrained language models often generate outputs that are not in line with human preferences, such as harmful text or factually incorrect summaries. Recent work approaches the above issues by learning from a simple form of human feedback: comparisons between pairs of model-generated outputs. However, comparison feedback only conveys limited information about human preferences. In this paper, we introduce Imitation learning from Language Feedback (ILF), a new approach that utilizes more informative language feedback. ILF consists of three steps that are applied iteratively: first, conditioning"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.16755","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-03-28T17:04:15Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"67cc2efdb93ee5fb6290af2923ea91c84cf4ff1af77b5f84f555cbe11531348c","abstract_canon_sha256":"0082c50fab6b6a9c70fd7d2110c5e2d2f5202514ef8aecfb17d687a5856b9c09"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:48:16.053780Z","signature_b64":"2Egg9QqRs+TceQuTfx8bYo23AhP6lKeufvC0LdDarQxLLPFKEZEEmMPGPPV3uHPQboXXRRDn6ekqEz3uMTTNCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"098a0e2051b1da6601447ff464163fd0e66114336f87e9e455fea53803ef473d","last_reissued_at":"2026-07-05T07:48:16.053285Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:48:16.053285Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Language Models with Language Feedback at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Angelica Chen, Ethan Perez, J\\'er\\'emy Scheurer, Jon Ander Campos, Jun Shern Chan, Kyunghyun Cho, Tomasz Korbak","submitted_at":"2023-03-28T17:04:15Z","abstract_excerpt":"Pretrained language models often generate outputs that are not in line with human preferences, such as harmful text or factually incorrect summaries. Recent work approaches the above issues by learning from a simple form of human feedback: comparisons between pairs of model-generated outputs. However, comparison feedback only conveys limited information about human preferences. In this paper, we introduce Imitation learning from Language Feedback (ILF), a new approach that utilizes more informative language feedback. ILF consists of three steps that are applied iteratively: first, conditioning"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.16755","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.16755/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.16755","created_at":"2026-07-05T07:48:16.053350+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.16755v3","created_at":"2026-07-05T07:48:16.053350+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.16755","created_at":"2026-07-05T07:48:16.053350+00:00"},{"alias_kind":"pith_short_12","alias_value":"BGFA4ICRWHNG","created_at":"2026-07-05T07:48:16.053350+00:00"},{"alias_kind":"pith_short_16","alias_value":"BGFA4ICRWHNGMAKE","created_at":"2026-07-05T07:48:16.053350+00:00"},{"alias_kind":"pith_short_8","alias_value":"BGFA4ICR","created_at":"2026-07-05T07:48:16.053350+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24547","citing_title":"RL with Learnable Textual Feedback: A Bilevel Approach","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20506","citing_title":"Reinforcing Human Behavior Simulation via Verbal Feedback","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15113","citing_title":"Learning from Language Feedback via Variational Policy Distillation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15224","citing_title":"ICRL: Learning to Internalize Self-Critique with Reinforcement Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2304.06767","citing_title":"RAFT: Reward rAnked FineTuning for Generative Foundation Model Alignment","ref_index":132,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BGFA4ICRWHNGMAKEP72GIFR72D","json":"https://pith.science/pith/BGFA4ICRWHNGMAKEP72GIFR72D.json","graph_json":"https://pith.science/api/pith-number/BGFA4ICRWHNGMAKEP72GIFR72D/graph.json","events_json":"https://pith.science/api/pith-number/BGFA4ICRWHNGMAKEP72GIFR72D/events.json","paper":"https://pith.science/paper/BGFA4ICR"},"agent_actions":{"view_html":"https://pith.science/pith/BGFA4ICRWHNGMAKEP72GIFR72D","download_json":"https://pith.science/pith/BGFA4ICRWHNGMAKEP72GIFR72D.json","view_paper":"https://pith.science/paper/BGFA4ICR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.16755&json=true","fetch_graph":"https://pith.science/api/pith-number/BGFA4ICRWHNGMAKEP72GIFR72D/graph.json","fetch_events":"https://pith.science/api/pith-number/BGFA4ICRWHNGMAKEP72GIFR72D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BGFA4ICRWHNGMAKEP72GIFR72D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BGFA4ICRWHNGMAKEP72GIFR72D/action/storage_attestation","attest_author":"https://pith.science/pith/BGFA4ICRWHNGMAKEP72GIFR72D/action/author_attestation","sign_citation":"https://pith.science/pith/BGFA4ICRWHNGMAKEP72GIFR72D/action/citation_signature","submit_replication":"https://pith.science/pith/BGFA4ICRWHNGMAKEP72GIFR72D/action/replication_record"}},"created_at":"2026-07-05T07:48:16.053350+00:00","updated_at":"2026-07-05T07:48:16.053350+00:00"}