{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2EDAIS7SA5W3LMS7PB75H7KDLE","short_pith_number":"pith:2EDAIS7S","schema_version":"1.0","canonical_sha256":"d106044bf2076db5b25f787fd3fd43591ddcd6fcefe31afff5d139256e0f6ab4","source":{"kind":"arxiv","id":"2402.05369","version":3},"attestation_state":"computed","paper":{"title":"Noise Contrastive Alignment of Language Models with Explicit Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ganqu Cui, Guande He, Hang Su, Huayu Chen, Jun Zhu, Lifan Yuan","submitted_at":"2024-02-08T02:58:47Z","abstract_excerpt":"User intentions are typically formalized as evaluation rewards to be maximized when fine-tuning language models (LMs). Existing alignment methods, such as Direct Preference Optimization (DPO), are mainly tailored for pairwise preference data where rewards are implicitly defined rather than explicitly given. In this paper, we introduce a general framework for LM alignment, leveraging Noise Contrastive Estimation (NCE) to bridge the gap in handling reward datasets explicitly annotated with scalar evaluations. Our framework comprises two parallel algorithms, NCA and InfoNCA, both enabling the dir"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05369","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-08T02:58:47Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"7a759a8b4487bc9306dca062a177dfe7fc0d4df6443d583b95b868a1b85fddab","abstract_canon_sha256":"dbc51c367b9520fa094531b45730fa4954f18e3efff4cfa543c22b3dc19040d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:28:10.473126Z","signature_b64":"wFDqEE4qAURWtZ7I0uTHL2xLUQZ+/1w+9Di6Es5g57cBKXklqUSxnSqPW93GzYVB3vk/gvfBrnZGLXCrHtfMDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d106044bf2076db5b25f787fd3fd43591ddcd6fcefe31afff5d139256e0f6ab4","last_reissued_at":"2026-07-05T09:28:10.472695Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:28:10.472695Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Noise Contrastive Alignment of Language Models with Explicit Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ganqu Cui, Guande He, Hang Su, Huayu Chen, Jun Zhu, Lifan Yuan","submitted_at":"2024-02-08T02:58:47Z","abstract_excerpt":"User intentions are typically formalized as evaluation rewards to be maximized when fine-tuning language models (LMs). Existing alignment methods, such as Direct Preference Optimization (DPO), are mainly tailored for pairwise preference data where rewards are implicitly defined rather than explicitly given. In this paper, we introduce a general framework for LM alignment, leveraging Noise Contrastive Estimation (NCE) to bridge the gap in handling reward datasets explicitly annotated with scalar evaluations. Our framework comprises two parallel algorithms, NCA and InfoNCA, both enabling the dir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05369","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05369/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05369","created_at":"2026-07-05T09:28:10.472752+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05369v3","created_at":"2026-07-05T09:28:10.472752+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05369","created_at":"2026-07-05T09:28:10.472752+00:00"},{"alias_kind":"pith_short_12","alias_value":"2EDAIS7SA5W3","created_at":"2026-07-05T09:28:10.472752+00:00"},{"alias_kind":"pith_short_16","alias_value":"2EDAIS7SA5W3LMS7","created_at":"2026-07-05T09:28:10.472752+00:00"},{"alias_kind":"pith_short_8","alias_value":"2EDAIS7S","created_at":"2026-07-05T09:28:10.472752+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24937","citing_title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","ref_index":193,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10442","citing_title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10622","citing_title":"Vocabulary Hijacking in LVLMs: Unveiling Critical Attention Heads by Excluding Inert Tokens to Mitigate Hallucination","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24536","citing_title":"Generating Place-Based Compromises Between Two Points of View","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2502.01456","citing_title":"Process Reinforcement through Implicit Rewards","ref_index":120,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2EDAIS7SA5W3LMS7PB75H7KDLE","json":"https://pith.science/pith/2EDAIS7SA5W3LMS7PB75H7KDLE.json","graph_json":"https://pith.science/api/pith-number/2EDAIS7SA5W3LMS7PB75H7KDLE/graph.json","events_json":"https://pith.science/api/pith-number/2EDAIS7SA5W3LMS7PB75H7KDLE/events.json","paper":"https://pith.science/paper/2EDAIS7S"},"agent_actions":{"view_html":"https://pith.science/pith/2EDAIS7SA5W3LMS7PB75H7KDLE","download_json":"https://pith.science/pith/2EDAIS7SA5W3LMS7PB75H7KDLE.json","view_paper":"https://pith.science/paper/2EDAIS7S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05369&json=true","fetch_graph":"https://pith.science/api/pith-number/2EDAIS7SA5W3LMS7PB75H7KDLE/graph.json","fetch_events":"https://pith.science/api/pith-number/2EDAIS7SA5W3LMS7PB75H7KDLE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2EDAIS7SA5W3LMS7PB75H7KDLE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2EDAIS7SA5W3LMS7PB75H7KDLE/action/storage_attestation","attest_author":"https://pith.science/pith/2EDAIS7SA5W3LMS7PB75H7KDLE/action/author_attestation","sign_citation":"https://pith.science/pith/2EDAIS7SA5W3LMS7PB75H7KDLE/action/citation_signature","submit_replication":"https://pith.science/pith/2EDAIS7SA5W3LMS7PB75H7KDLE/action/replication_record"}},"created_at":"2026-07-05T09:28:10.472752+00:00","updated_at":"2026-07-05T09:28:10.472752+00:00"}