{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:PMVQPIL2OD3CIMYDZUJND2I4GO","short_pith_number":"pith:PMVQPIL2","schema_version":"1.0","canonical_sha256":"7b2b07a17a70f6243303cd12d1e91c339f272b50d6651a412fb24a8624a34329","source":{"kind":"arxiv","id":"2204.14146","version":4},"attestation_state":"computed","paper":{"title":"Training Language Models with Language Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Angelica Chen, Ethan Perez, J\\'er\\'emy Scheurer, Jon Ander Campos, Jun Shern Chan, Kyunghyun Cho","submitted_at":"2022-04-29T15:06:58Z","abstract_excerpt":"Pretrained language models often do not perform tasks in ways that are in line with our preferences, e.g., generating offensive text or factually incorrect summaries. Recent work approaches the above issue by learning from a simple form of human evaluation: comparisons between pairs of model-generated task outputs. Comparison feedback conveys limited information about human preferences per human evaluation. Here, we propose to learn from natural language feedback, which conveys more information per human evaluation. We learn from language feedback on model outputs using a three-step learning a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.14146","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-04-29T15:06:58Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f636a12d18e92c8dc8c73d077f642da2d552c7b817902c9d0773b513dde36d41","abstract_canon_sha256":"35b8ac2c0782583e606db838a3786a6858ed893e51a220ba660ee5839d2335fd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:17:01.867828Z","signature_b64":"UCWbUl+fU3yX0EsMzjkq3S52mGxxj7gxjTz0+GoiNy5i/eE8fxKaTDnaBCNveTzIPsb+fwseUB4J2FD4OcZNAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b2b07a17a70f6243303cd12d1e91c339f272b50d6651a412fb24a8624a34329","last_reissued_at":"2026-07-05T05:17:01.867374Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:17:01.867374Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Language Models with Language Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Angelica Chen, Ethan Perez, J\\'er\\'emy Scheurer, Jon Ander Campos, Jun Shern Chan, Kyunghyun Cho","submitted_at":"2022-04-29T15:06:58Z","abstract_excerpt":"Pretrained language models often do not perform tasks in ways that are in line with our preferences, e.g., generating offensive text or factually incorrect summaries. Recent work approaches the above issue by learning from a simple form of human evaluation: comparisons between pairs of model-generated task outputs. Comparison feedback conveys limited information about human preferences per human evaluation. Here, we propose to learn from natural language feedback, which conveys more information per human evaluation. We learn from language feedback on model outputs using a three-step learning a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.14146","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.14146/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.14146","created_at":"2026-07-05T05:17:01.867431+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.14146v4","created_at":"2026-07-05T05:17:01.867431+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.14146","created_at":"2026-07-05T05:17:01.867431+00:00"},{"alias_kind":"pith_short_12","alias_value":"PMVQPIL2OD3C","created_at":"2026-07-05T05:17:01.867431+00:00"},{"alias_kind":"pith_short_16","alias_value":"PMVQPIL2OD3CIMYD","created_at":"2026-07-05T05:17:01.867431+00:00"},{"alias_kind":"pith_short_8","alias_value":"PMVQPIL2","created_at":"2026-07-05T05:17:01.867431+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04273","citing_title":"Characterizing initial human-AI proof formalization workflows","ref_index":298,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03715","citing_title":"R$^3$L: Reflect-then-Retry Reinforcement Learning with Language-Guided Exploration, Pivotal Credit, and Positive Amplification","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2302.12192","citing_title":"Aligning Text-to-Image Models using Human Feedback","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2303.17651","citing_title":"Self-Refine: Iterative Refinement with Self-Feedback","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PMVQPIL2OD3CIMYDZUJND2I4GO","json":"https://pith.science/pith/PMVQPIL2OD3CIMYDZUJND2I4GO.json","graph_json":"https://pith.science/api/pith-number/PMVQPIL2OD3CIMYDZUJND2I4GO/graph.json","events_json":"https://pith.science/api/pith-number/PMVQPIL2OD3CIMYDZUJND2I4GO/events.json","paper":"https://pith.science/paper/PMVQPIL2"},"agent_actions":{"view_html":"https://pith.science/pith/PMVQPIL2OD3CIMYDZUJND2I4GO","download_json":"https://pith.science/pith/PMVQPIL2OD3CIMYDZUJND2I4GO.json","view_paper":"https://pith.science/paper/PMVQPIL2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.14146&json=true","fetch_graph":"https://pith.science/api/pith-number/PMVQPIL2OD3CIMYDZUJND2I4GO/graph.json","fetch_events":"https://pith.science/api/pith-number/PMVQPIL2OD3CIMYDZUJND2I4GO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PMVQPIL2OD3CIMYDZUJND2I4GO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PMVQPIL2OD3CIMYDZUJND2I4GO/action/storage_attestation","attest_author":"https://pith.science/pith/PMVQPIL2OD3CIMYDZUJND2I4GO/action/author_attestation","sign_citation":"https://pith.science/pith/PMVQPIL2OD3CIMYDZUJND2I4GO/action/citation_signature","submit_replication":"https://pith.science/pith/PMVQPIL2OD3CIMYDZUJND2I4GO/action/replication_record"}},"created_at":"2026-07-05T05:17:01.867431+00:00","updated_at":"2026-07-05T05:17:01.867431+00:00"}