{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VJ4I5GYPUOI53I2WVGDV6ZJ3TK","short_pith_number":"pith:VJ4I5GYP","schema_version":"1.0","canonical_sha256":"aa788e9b0fa391dda356a9875f653b9a889ef6afbc845339768248761f7a717b","source":{"kind":"arxiv","id":"2305.10425","version":1},"attestation_state":"computed","paper":{"title":"SLiC-HF: Sequence Likelihood Calibration with Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Misha Khalman, Mohammad Saleh, Peter J. Liu, Rishabh Joshi, Tianqi Liu, Yao Zhao","submitted_at":"2023-05-17T17:57:10Z","abstract_excerpt":"Learning from human feedback has been shown to be effective at aligning language models with human preferences. Past work has often relied on Reinforcement Learning from Human Feedback (RLHF), which optimizes the language model using reward scores assigned from a reward model trained on human preference data. In this work we show how the recently introduced Sequence Likelihood Calibration (SLiC), can also be used to effectively learn from human preferences (SLiC-HF). Furthermore, we demonstrate this can be done with human feedback data collected for a different model, similar to off-policy, of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.10425","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-17T17:57:10Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c9936b4b9dd69545e9d57c36edcfa92fb9ee754f045937b79fe4495b323f7ef6","abstract_canon_sha256":"116be9e20f6ec229e2f914482a223025774a8175a4563e2b31efcfd75783e5c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:11:14.347344Z","signature_b64":"Jh+6r6AhP8xn1EO45Nm/f8rB+sDM2YXkz+eU0uDoLbZEMXw5EwCJdzMLr4Gsrbkr6RTN5j8Qs1KbvbHMwEhGDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aa788e9b0fa391dda356a9875f653b9a889ef6afbc845339768248761f7a717b","last_reissued_at":"2026-07-05T06:11:14.346814Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:11:14.346814Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SLiC-HF: Sequence Likelihood Calibration with Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Misha Khalman, Mohammad Saleh, Peter J. Liu, Rishabh Joshi, Tianqi Liu, Yao Zhao","submitted_at":"2023-05-17T17:57:10Z","abstract_excerpt":"Learning from human feedback has been shown to be effective at aligning language models with human preferences. Past work has often relied on Reinforcement Learning from Human Feedback (RLHF), which optimizes the language model using reward scores assigned from a reward model trained on human preference data. In this work we show how the recently introduced Sequence Likelihood Calibration (SLiC), can also be used to effectively learn from human preferences (SLiC-HF). Furthermore, we demonstrate this can be done with human feedback data collected for a different model, similar to off-policy, of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.10425","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.10425/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.10425","created_at":"2026-07-05T06:11:14.346878+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.10425v1","created_at":"2026-07-05T06:11:14.346878+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.10425","created_at":"2026-07-05T06:11:14.346878+00:00"},{"alias_kind":"pith_short_12","alias_value":"VJ4I5GYPUOI5","created_at":"2026-07-05T06:11:14.346878+00:00"},{"alias_kind":"pith_short_16","alias_value":"VJ4I5GYPUOI53I2W","created_at":"2026-07-05T06:11:14.346878+00:00"},{"alias_kind":"pith_short_8","alias_value":"VJ4I5GYP","created_at":"2026-07-05T06:11:14.346878+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":28,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24937","citing_title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","ref_index":194,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13006","citing_title":"Emo-LiPO: Listwise Preference Optimization for Fine-Grained Emotion Intensity Control in LLM-based Text-to-Speech","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12578","citing_title":"MARD: Mirror-Augmented Reasoning Distillation for Mechanism-Level Drug-Drug Interaction Prediction","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03376","citing_title":"P$^2$-DPO: Grounding Hallucination in Perceptual Processing via Calibration Direct Preference Optimization","ref_index":119,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01561","citing_title":"S-SPPO: Semantic-Calibrated Self-Play Preference Optimization","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28401","citing_title":"Vision-driven Preference Synthesis for Mitigating Hallucinations in VLMs","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02439","citing_title":"Anomaly-Preference Image Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04477","citing_title":"Data-dependent Exploration for Online Reinforcement Learning from Human Feedback","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30248","citing_title":"Your Data Manifold is Secretly a Reward Model: Shell-LCC for Text-to-Video Generation","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2412.02125","citing_title":"Preference Goal Tuning: Post-Training as Latent Control for Frozen Policies","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2502.06387","citing_title":"How Humans Help LLMs: Assessing and Incentivizing Human Preference Annotators","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2504.12501","citing_title":"Reinforcement Learning from Human Feedback","ref_index":193,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21883","citing_title":"Token-weighted Direct Preference Optimization with Attention","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20834","citing_title":"Conditional Equivalence of DPO and RLHF: Implicit Assumption, Failure Modes, and Provable Alignment","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02439","citing_title":"Anomaly-Preference Image Generation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2408.07199","citing_title":"Agent Q: Advanced Reasoning and Learning for Autonomous AI Agents","ref_index":210,"is_internal_anchor":false},{"citing_arxiv_id":"2505.19134","citing_title":"Incentivizing High-Quality Human Annotations with Golden Questions","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02850","citing_title":"LLM Hypnosis: Exploiting User Feedback for Unauthorized Knowledge Injection to All Users","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20265","citing_title":"Failure Modes of Maximum Entropy RLHF","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2408.00724","citing_title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","ref_index":237,"is_internal_anchor":false},{"citing_arxiv_id":"2510.17881","citing_title":"POPI: Personalizing LLMs via Optimized Natural Language Preference Inference","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12545","citing_title":"CROP: Expert-Aligned Image Cropping via Compositional Reasoning and Optimizing Preference","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2401.10020","citing_title":"Self-Rewarding Language Models","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2402.01306","citing_title":"KTO: Model Alignment as Prospect Theoretic Optimization","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27733","citing_title":"Mind the Gap: Structure-Aware Consistency in Preference Learning","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VJ4I5GYPUOI53I2WVGDV6ZJ3TK","json":"https://pith.science/pith/VJ4I5GYPUOI53I2WVGDV6ZJ3TK.json","graph_json":"https://pith.science/api/pith-number/VJ4I5GYPUOI53I2WVGDV6ZJ3TK/graph.json","events_json":"https://pith.science/api/pith-number/VJ4I5GYPUOI53I2WVGDV6ZJ3TK/events.json","paper":"https://pith.science/paper/VJ4I5GYP"},"agent_actions":{"view_html":"https://pith.science/pith/VJ4I5GYPUOI53I2WVGDV6ZJ3TK","download_json":"https://pith.science/pith/VJ4I5GYPUOI53I2WVGDV6ZJ3TK.json","view_paper":"https://pith.science/paper/VJ4I5GYP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.10425&json=true","fetch_graph":"https://pith.science/api/pith-number/VJ4I5GYPUOI53I2WVGDV6ZJ3TK/graph.json","fetch_events":"https://pith.science/api/pith-number/VJ4I5GYPUOI53I2WVGDV6ZJ3TK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VJ4I5GYPUOI53I2WVGDV6ZJ3TK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VJ4I5GYPUOI53I2WVGDV6ZJ3TK/action/storage_attestation","attest_author":"https://pith.science/pith/VJ4I5GYPUOI53I2WVGDV6ZJ3TK/action/author_attestation","sign_citation":"https://pith.science/pith/VJ4I5GYPUOI53I2WVGDV6ZJ3TK/action/citation_signature","submit_replication":"https://pith.science/pith/VJ4I5GYPUOI53I2WVGDV6ZJ3TK/action/replication_record"}},"created_at":"2026-07-05T06:11:14.346878+00:00","updated_at":"2026-07-05T06:11:14.346878+00:00"}