{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IYQGHBUF7YTZNHSNKSIXFS32NB","short_pith_number":"pith:IYQGHBUF","schema_version":"1.0","canonical_sha256":"4620638685fe27969e4d549172cb7a684f11beff1cdc6e58d2d48a8a1732188f","source":{"kind":"arxiv","id":"2302.02676","version":8},"attestation_state":"computed","paper":{"title":"Chain of Hindsight Aligns Language Models with Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Carmelo Sferrazza, Hao Liu, Pieter Abbeel","submitted_at":"2023-02-06T10:28:16Z","abstract_excerpt":"Learning from human preferences is important for language models to match human needs and to align with human and social values. Prior works have achieved remarkable successes by learning from human feedback to understand and follow instructions. Nonetheless, these methods are either founded on hand-picked model generations that are favored by human annotators, rendering them inefficient in terms of data utilization and challenging to apply in general, or they depend on reinforcement learning, which often suffers from imperfect reward functions and relies on extremely challenging optimizations"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.02676","kind":"arxiv","version":8},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-02-06T10:28:16Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"7ab42bb991ad46ff93723cfe4fb6832fc0c1dd5108d87e43eae55b0afdc61781","abstract_canon_sha256":"40b167e8d456da121b240c068b29e7e900df2cc3d46b6246d57ecbd48329d9df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:58.610207Z","signature_b64":"XgQYS7kn+bJtlCQi6hiFgFrGCj1W/g1VE4UgmXAsyK8m90tdDCTfFp5cXrfHHLKPOxGiTiw8x/LTSBfgdrqEDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4620638685fe27969e4d549172cb7a684f11beff1cdc6e58d2d48a8a1732188f","last_reissued_at":"2026-07-05T07:01:58.609705Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:58.609705Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Chain of Hindsight Aligns Language Models with Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Carmelo Sferrazza, Hao Liu, Pieter Abbeel","submitted_at":"2023-02-06T10:28:16Z","abstract_excerpt":"Learning from human preferences is important for language models to match human needs and to align with human and social values. Prior works have achieved remarkable successes by learning from human feedback to understand and follow instructions. Nonetheless, these methods are either founded on hand-picked model generations that are favored by human annotators, rendering them inefficient in terms of data utilization and challenging to apply in general, or they depend on reinforcement learning, which often suffers from imperfect reward functions and relies on extremely challenging optimizations"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.02676","kind":"arxiv","version":8},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.02676/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.02676","created_at":"2026-07-05T07:01:58.609771+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.02676v8","created_at":"2026-07-05T07:01:58.609771+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.02676","created_at":"2026-07-05T07:01:58.609771+00:00"},{"alias_kind":"pith_short_12","alias_value":"IYQGHBUF7YTZ","created_at":"2026-07-05T07:01:58.609771+00:00"},{"alias_kind":"pith_short_16","alias_value":"IYQGHBUF7YTZNHSN","created_at":"2026-07-05T07:01:58.609771+00:00"},{"alias_kind":"pith_short_8","alias_value":"IYQGHBUF","created_at":"2026-07-05T07:01:58.609771+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02293","citing_title":"AI as a Tool for Simulation-Based Experiments in Literary Studies","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2501.16150","citing_title":"A Comprehensive Survey of Agents for Computer Use: Foundations, Challenges, and Future Directions","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20506","citing_title":"Reinforcing Human Behavior Simulation via Verbal Feedback","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15113","citing_title":"Learning from Language Feedback via Variational Policy Distillation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":172,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05561","citing_title":"TrustLLM: Trustworthiness in Large Language Models","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2412.13171","citing_title":"Compressed Chain of Thought: Efficient Reasoning Through Dense Representations","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2310.06987","citing_title":"Catastrophic Jailbreak of Open-source LLMs via Exploiting Generation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2302.12192","citing_title":"Aligning Text-to-Image Models using Human Feedback","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2304.05128","citing_title":"Teaching Large Language Models to Self-Debug","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2309.07864","citing_title":"The Rise and Potential of Large Language Model Based Agents: A Survey","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":142,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15136","citing_title":"Feedback-Driven Execution for LLM-Based Binary Analysis","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IYQGHBUF7YTZNHSNKSIXFS32NB","json":"https://pith.science/pith/IYQGHBUF7YTZNHSNKSIXFS32NB.json","graph_json":"https://pith.science/api/pith-number/IYQGHBUF7YTZNHSNKSIXFS32NB/graph.json","events_json":"https://pith.science/api/pith-number/IYQGHBUF7YTZNHSNKSIXFS32NB/events.json","paper":"https://pith.science/paper/IYQGHBUF"},"agent_actions":{"view_html":"https://pith.science/pith/IYQGHBUF7YTZNHSNKSIXFS32NB","download_json":"https://pith.science/pith/IYQGHBUF7YTZNHSNKSIXFS32NB.json","view_paper":"https://pith.science/paper/IYQGHBUF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.02676&json=true","fetch_graph":"https://pith.science/api/pith-number/IYQGHBUF7YTZNHSNKSIXFS32NB/graph.json","fetch_events":"https://pith.science/api/pith-number/IYQGHBUF7YTZNHSNKSIXFS32NB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IYQGHBUF7YTZNHSNKSIXFS32NB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IYQGHBUF7YTZNHSNKSIXFS32NB/action/storage_attestation","attest_author":"https://pith.science/pith/IYQGHBUF7YTZNHSNKSIXFS32NB/action/author_attestation","sign_citation":"https://pith.science/pith/IYQGHBUF7YTZNHSNKSIXFS32NB/action/citation_signature","submit_replication":"https://pith.science/pith/IYQGHBUF7YTZNHSNKSIXFS32NB/action/replication_record"}},"created_at":"2026-07-05T07:01:58.609771+00:00","updated_at":"2026-07-05T07:01:58.609771+00:00"}