{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SW5UKBBSO2KO2356JUN5OBGV5X","short_pith_number":"pith:SW5UKBBS","schema_version":"1.0","canonical_sha256":"95bb4504327694ed6fbe4d1bd704d5edc939d3467759247920fd258d5e0ea06b","source":{"kind":"arxiv","id":"2404.04656","version":2},"attestation_state":"computed","paper":{"title":"Binary Classifier Optimization for Large Language Model Alignment","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Daniel Wontae Nam, Gunsoo Han, Kyoung-Woon On, Seungjae Jung","submitted_at":"2024-04-06T15:20:59Z","abstract_excerpt":"In real-world services such as ChatGPT, aligning models based on user feedback is crucial for improving model performance. However, due to the simplicity and convenience of providing feedback, users typically offer only basic binary signals, such as 'thumbs-up' or 'thumbs-down'. Most existing alignment research, on the other hand, relies on preference-based approaches that require both positive and negative responses as a pair. We propose Binary Classifier Optimization (BCO), a technique that effectively aligns LLMs using only binary feedback. BCO trains a binary classifier, where the logit se"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.04656","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-04-06T15:20:59Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"ca2de6e19d74d5f32efd509386bc494542f1bbb79c5af97ffc9b731dfeb1467e","abstract_canon_sha256":"1c82ae14a158af9d429bf47d2e37359c0aa98e96f7df7dabc1e778b832529b6a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:11.973938Z","signature_b64":"a+mYIql0LtoK7gbLngm0nG5Rd3fKD13AAu3T7hN52iFufi4JnHO/rbyhoEy8yU/G489PIbaXxolFGG+zR0DYCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"95bb4504327694ed6fbe4d1bd704d5edc939d3467759247920fd258d5e0ea06b","last_reissued_at":"2026-07-05T11:18:11.973410Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:11.973410Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Binary Classifier Optimization for Large Language Model Alignment","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Daniel Wontae Nam, Gunsoo Han, Kyoung-Woon On, Seungjae Jung","submitted_at":"2024-04-06T15:20:59Z","abstract_excerpt":"In real-world services such as ChatGPT, aligning models based on user feedback is crucial for improving model performance. However, due to the simplicity and convenience of providing feedback, users typically offer only basic binary signals, such as 'thumbs-up' or 'thumbs-down'. Most existing alignment research, on the other hand, relies on preference-based approaches that require both positive and negative responses as a pair. We propose Binary Classifier Optimization (BCO), a technique that effectively aligns LLMs using only binary feedback. BCO trains a binary classifier, where the logit se"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.04656","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.04656/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.04656","created_at":"2026-07-05T11:18:11.973473+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.04656v2","created_at":"2026-07-05T11:18:11.973473+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.04656","created_at":"2026-07-05T11:18:11.973473+00:00"},{"alias_kind":"pith_short_12","alias_value":"SW5UKBBSO2KO","created_at":"2026-07-05T11:18:11.973473+00:00"},{"alias_kind":"pith_short_16","alias_value":"SW5UKBBSO2KO2356","created_at":"2026-07-05T11:18:11.973473+00:00"},{"alias_kind":"pith_short_8","alias_value":"SW5UKBBS","created_at":"2026-07-05T11:18:11.973473+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2411.10442","citing_title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11217","citing_title":"Leveraging RAG for Training-Free Alignment of LLMs","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04410","citing_title":"Relative Density Ratio Optimization for Stable and Statistically Consistent Model Alignment","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10479","citing_title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2508.18265","citing_title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SW5UKBBSO2KO2356JUN5OBGV5X","json":"https://pith.science/pith/SW5UKBBSO2KO2356JUN5OBGV5X.json","graph_json":"https://pith.science/api/pith-number/SW5UKBBSO2KO2356JUN5OBGV5X/graph.json","events_json":"https://pith.science/api/pith-number/SW5UKBBSO2KO2356JUN5OBGV5X/events.json","paper":"https://pith.science/paper/SW5UKBBS"},"agent_actions":{"view_html":"https://pith.science/pith/SW5UKBBSO2KO2356JUN5OBGV5X","download_json":"https://pith.science/pith/SW5UKBBSO2KO2356JUN5OBGV5X.json","view_paper":"https://pith.science/paper/SW5UKBBS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.04656&json=true","fetch_graph":"https://pith.science/api/pith-number/SW5UKBBSO2KO2356JUN5OBGV5X/graph.json","fetch_events":"https://pith.science/api/pith-number/SW5UKBBSO2KO2356JUN5OBGV5X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SW5UKBBSO2KO2356JUN5OBGV5X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SW5UKBBSO2KO2356JUN5OBGV5X/action/storage_attestation","attest_author":"https://pith.science/pith/SW5UKBBSO2KO2356JUN5OBGV5X/action/author_attestation","sign_citation":"https://pith.science/pith/SW5UKBBSO2KO2356JUN5OBGV5X/action/citation_signature","submit_replication":"https://pith.science/pith/SW5UKBBSO2KO2356JUN5OBGV5X/action/replication_record"}},"created_at":"2026-07-05T11:18:11.973473+00:00","updated_at":"2026-07-05T11:18:11.973473+00:00"}