{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:56QHQXUG7NEIOSOPMZ5JUH3SQF","short_pith_number":"pith:56QHQXUG","schema_version":"1.0","canonical_sha256":"efa0785e86fb488749cf667a9a1f72815419830bf79645103e6f09b1c076184d","source":{"kind":"arxiv","id":"2406.16061","version":2},"attestation_state":"computed","paper":{"title":"PORT: Preference Optimization on Reasoning Traces","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Abdalgader Abubaker, Hakim Hacid, Salem Lahlou","submitted_at":"2024-06-23T09:51:06Z","abstract_excerpt":"Preference optimization methods have been successfully applied to improve not only the alignment of large language models (LLMs) with human values, but also specific natural language tasks such as summarization and stylistic continuations. This paper proposes using preference optimization methods on Chain-of-Thought steps in order to improve the mathematical reasoning performances of language models. While the chosen answers are obtained from datasets that include reasoning traces, we propose two complementary schemes for generating rejected answers: weak LLM prompting, and digit corruption. O"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.16061","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-23T09:51:06Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"fc8a249c43d03a0b6a921d2593368701f626ed9910f38cb910667f64d9f142b6","abstract_canon_sha256":"7fa2e1f2507c2f08c2be083215936ce0a7155ffedbf4edbabbdd5aa17d14f65f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:45.021123Z","signature_b64":"DR0PDb1/er7gPk72r6n3nm+cYVfJbqPuGxTitCs5gCFABU6pFkgHGFZd89AghfMVnm2ehArJuM3KBxCwHonLCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"efa0785e86fb488749cf667a9a1f72815419830bf79645103e6f09b1c076184d","last_reissued_at":"2026-07-05T10:09:45.020632Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:45.020632Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PORT: Preference Optimization on Reasoning Traces","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Abdalgader Abubaker, Hakim Hacid, Salem Lahlou","submitted_at":"2024-06-23T09:51:06Z","abstract_excerpt":"Preference optimization methods have been successfully applied to improve not only the alignment of large language models (LLMs) with human values, but also specific natural language tasks such as summarization and stylistic continuations. This paper proposes using preference optimization methods on Chain-of-Thought steps in order to improve the mathematical reasoning performances of language models. While the chosen answers are obtained from datasets that include reasoning traces, we propose two complementary schemes for generating rejected answers: weak LLM prompting, and digit corruption. O"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16061","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.16061/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.16061","created_at":"2026-07-05T10:09:45.020690+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.16061v2","created_at":"2026-07-05T10:09:45.020690+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16061","created_at":"2026-07-05T10:09:45.020690+00:00"},{"alias_kind":"pith_short_12","alias_value":"56QHQXUG7NEI","created_at":"2026-07-05T10:09:45.020690+00:00"},{"alias_kind":"pith_short_16","alias_value":"56QHQXUG7NEIOSOP","created_at":"2026-07-05T10:09:45.020690+00:00"},{"alias_kind":"pith_short_8","alias_value":"56QHQXUG","created_at":"2026-07-05T10:09:45.020690+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.02726","citing_title":"RACE-Align: Retrieval-Augmented and Chain-of-Thought Enhanced Preference Alignment for Large Language Models","ref_index":43,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/56QHQXUG7NEIOSOPMZ5JUH3SQF","json":"https://pith.science/pith/56QHQXUG7NEIOSOPMZ5JUH3SQF.json","graph_json":"https://pith.science/api/pith-number/56QHQXUG7NEIOSOPMZ5JUH3SQF/graph.json","events_json":"https://pith.science/api/pith-number/56QHQXUG7NEIOSOPMZ5JUH3SQF/events.json","paper":"https://pith.science/paper/56QHQXUG"},"agent_actions":{"view_html":"https://pith.science/pith/56QHQXUG7NEIOSOPMZ5JUH3SQF","download_json":"https://pith.science/pith/56QHQXUG7NEIOSOPMZ5JUH3SQF.json","view_paper":"https://pith.science/paper/56QHQXUG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.16061&json=true","fetch_graph":"https://pith.science/api/pith-number/56QHQXUG7NEIOSOPMZ5JUH3SQF/graph.json","fetch_events":"https://pith.science/api/pith-number/56QHQXUG7NEIOSOPMZ5JUH3SQF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/56QHQXUG7NEIOSOPMZ5JUH3SQF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/56QHQXUG7NEIOSOPMZ5JUH3SQF/action/storage_attestation","attest_author":"https://pith.science/pith/56QHQXUG7NEIOSOPMZ5JUH3SQF/action/author_attestation","sign_citation":"https://pith.science/pith/56QHQXUG7NEIOSOPMZ5JUH3SQF/action/citation_signature","submit_replication":"https://pith.science/pith/56QHQXUG7NEIOSOPMZ5JUH3SQF/action/replication_record"}},"created_at":"2026-07-05T10:09:45.020690+00:00","updated_at":"2026-07-05T10:09:45.020690+00:00"}