{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RMBKLCLBP3KOI7HAPHJSAZ2XPT","short_pith_number":"pith:RMBKLCLB","schema_version":"1.0","canonical_sha256":"8b02a589617ed4e47ce079d32067577ceb5dbe3fc8f6534762cda423d1b40817","source":{"kind":"arxiv","id":"2411.04109","version":3},"attestation_state":"computed","paper":{"title":"Self-Consistency Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Archiki Prasad, Jane Yu, Jason Weston, Jing Xu, Maryam Fazel-Zarandi, Mohit Bansal, Richard Yuanzhe Pang, Sainbayar Sukhbaatar, Weizhe Yuan","submitted_at":"2024-11-06T18:36:22Z","abstract_excerpt":"Self-alignment, whereby models learn to improve themselves without human annotation, is a rapidly growing research area. However, existing techniques often fail to improve complex reasoning tasks due to the difficulty of assigning correct rewards. An orthogonal approach that is known to improve correctness is self-consistency, a method applied at inference time based on multiple sampling in order to find the most consistent answer. In this work, we extend the self-consistency concept to help train models. We thus introduce self-consistency preference optimization (ScPO), which iteratively trai"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.04109","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-06T18:36:22Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b5d37663db8810356d7e7f3633d2aeb90ed6008ea9d886cfe27ce1bac5f230c4","abstract_canon_sha256":"4ddd3cfb4e9c1f335f4ece327a2dda01b3dca1b8547f189c9ed83ea4a7afe3cc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:32:20.539293Z","signature_b64":"i3WJ1l3CjmfuLsNld6WQHc7zpQag2VPohJy3RtAfaRk22vwrzLBvwqqOFfPGUNZU40FZZzKcr5Oqa7cmY5nRDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b02a589617ed4e47ce079d32067577ceb5dbe3fc8f6534762cda423d1b40817","last_reissued_at":"2026-07-05T11:32:20.538710Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:32:20.538710Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Consistency Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Archiki Prasad, Jane Yu, Jason Weston, Jing Xu, Maryam Fazel-Zarandi, Mohit Bansal, Richard Yuanzhe Pang, Sainbayar Sukhbaatar, Weizhe Yuan","submitted_at":"2024-11-06T18:36:22Z","abstract_excerpt":"Self-alignment, whereby models learn to improve themselves without human annotation, is a rapidly growing research area. However, existing techniques often fail to improve complex reasoning tasks due to the difficulty of assigning correct rewards. An orthogonal approach that is known to improve correctness is self-consistency, a method applied at inference time based on multiple sampling in order to find the most consistent answer. In this work, we extend the self-consistency concept to help train models. We thus introduce self-consistency preference optimization (ScPO), which iteratively trai"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.04109","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.04109/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.04109","created_at":"2026-07-05T11:32:20.538786+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.04109v3","created_at":"2026-07-05T11:32:20.538786+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.04109","created_at":"2026-07-05T11:32:20.538786+00:00"},{"alias_kind":"pith_short_12","alias_value":"RMBKLCLBP3KO","created_at":"2026-07-05T11:32:20.538786+00:00"},{"alias_kind":"pith_short_16","alias_value":"RMBKLCLBP3KOI7HA","created_at":"2026-07-05T11:32:20.538786+00:00"},{"alias_kind":"pith_short_8","alias_value":"RMBKLCLB","created_at":"2026-07-05T11:32:20.538786+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18903","citing_title":"Reasoning Portability: Guiding Continual Learning for MLLMs in the RLVR Era","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15207","citing_title":"TeamTR: Trust-Region Fine-Tuning for Multi-Agent LLM Coordination","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01194","citing_title":"VLA-ATTC: Adaptive Test-Time Compute for VLA Models with Relative Action Critic Model","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04065","citing_title":"Free Energy-Driven Reinforcement Learning with Adaptive Advantage Shaping for Unsupervised Reasoning in LLMs","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RMBKLCLBP3KOI7HAPHJSAZ2XPT","json":"https://pith.science/pith/RMBKLCLBP3KOI7HAPHJSAZ2XPT.json","graph_json":"https://pith.science/api/pith-number/RMBKLCLBP3KOI7HAPHJSAZ2XPT/graph.json","events_json":"https://pith.science/api/pith-number/RMBKLCLBP3KOI7HAPHJSAZ2XPT/events.json","paper":"https://pith.science/paper/RMBKLCLB"},"agent_actions":{"view_html":"https://pith.science/pith/RMBKLCLBP3KOI7HAPHJSAZ2XPT","download_json":"https://pith.science/pith/RMBKLCLBP3KOI7HAPHJSAZ2XPT.json","view_paper":"https://pith.science/paper/RMBKLCLB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.04109&json=true","fetch_graph":"https://pith.science/api/pith-number/RMBKLCLBP3KOI7HAPHJSAZ2XPT/graph.json","fetch_events":"https://pith.science/api/pith-number/RMBKLCLBP3KOI7HAPHJSAZ2XPT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RMBKLCLBP3KOI7HAPHJSAZ2XPT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RMBKLCLBP3KOI7HAPHJSAZ2XPT/action/storage_attestation","attest_author":"https://pith.science/pith/RMBKLCLBP3KOI7HAPHJSAZ2XPT/action/author_attestation","sign_citation":"https://pith.science/pith/RMBKLCLBP3KOI7HAPHJSAZ2XPT/action/citation_signature","submit_replication":"https://pith.science/pith/RMBKLCLBP3KOI7HAPHJSAZ2XPT/action/replication_record"}},"created_at":"2026-07-05T11:32:20.538786+00:00","updated_at":"2026-07-05T11:32:20.538786+00:00"}