{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:SAASNGDQHDTBCST3UVG6DOMNUK","short_pith_number":"pith:SAASNGDQ","canonical_record":{"source":{"id":"2411.16345","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-25T12:44:02Z","cross_cats_sorted":[],"title_canon_sha256":"81595d98840b2bd133d9d6fdab122236bd1baa0c37f03d7f42641a5b5d765081","abstract_canon_sha256":"c0574412136553627d35b95e67c68cdec371a566d4b66d14d248034047ca720e"},"schema_version":"1.0"},"canonical_sha256":"900126987038e6114a7ba54de1b98da299cb9117a987fa3aada5344f4c50fc69","source":{"kind":"arxiv","id":"2411.16345","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2411.16345","created_at":"2026-07-05T10:14:07Z"},{"alias_kind":"arxiv_version","alias_value":"2411.16345v2","created_at":"2026-07-05T10:14:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.16345","created_at":"2026-07-05T10:14:07Z"},{"alias_kind":"pith_short_12","alias_value":"SAASNGDQHDTB","created_at":"2026-07-05T10:14:07Z"},{"alias_kind":"pith_short_16","alias_value":"SAASNGDQHDTBCST3","created_at":"2026-07-05T10:14:07Z"},{"alias_kind":"pith_short_8","alias_value":"SAASNGDQ","created_at":"2026-07-05T10:14:07Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:SAASNGDQHDTBCST3UVG6DOMNUK","target":"record","payload":{"canonical_record":{"source":{"id":"2411.16345","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-25T12:44:02Z","cross_cats_sorted":[],"title_canon_sha256":"81595d98840b2bd133d9d6fdab122236bd1baa0c37f03d7f42641a5b5d765081","abstract_canon_sha256":"c0574412136553627d35b95e67c68cdec371a566d4b66d14d248034047ca720e"},"schema_version":"1.0"},"canonical_sha256":"900126987038e6114a7ba54de1b98da299cb9117a987fa3aada5344f4c50fc69","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:07.212952Z","signature_b64":"Wzu+7ZVPhl4PPnvefVzMuSIizdYtQa68K0YG4+jHwsx8NuazSXkyPxMtuf0L4t88xaJA453jXWabgz0yZ1iKDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"900126987038e6114a7ba54de1b98da299cb9117a987fa3aada5344f4c50fc69","last_reissued_at":"2026-07-05T10:14:07.212464Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:07.212464Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2411.16345","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:14:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"BpsNzmaBSYqmQDZcRcmfYWzrHlYDkoT2WP9P37Un++xiZCSbMzeNcHs4le/zM7KXM2usyUgj2T3KG0+WywmEAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T08:13:01.647800Z"},"content_sha256":"a3ee9ab94f49ed13cbac9d8b3ec9dea4b8782209a6b549eb1623c74cf5aa0d2c","schema_version":"1.0","event_id":"sha256:a3ee9ab94f49ed13cbac9d8b3ec9dea4b8782209a6b549eb1623c74cf5aa0d2c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:SAASNGDQHDTBCST3UVG6DOMNUK","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Preference Optimization for Reasoning with Pseudo Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fangkai Jiao, Furu Wei, Geyang Guo, Nancy F. Chen, Shafiq Joty, Xingxing Zhang","submitted_at":"2024-11-25T12:44:02Z","abstract_excerpt":"Preference optimization techniques, such as Direct Preference Optimization (DPO), are frequently employed to enhance the reasoning capabilities of large language models (LLMs) in domains like mathematical reasoning and coding, typically following supervised fine-tuning. These methods rely on high-quality labels for reasoning tasks to generate preference pairs; however, the availability of reasoning datasets with human-verified labels is limited. In this study, we introduce a novel approach to generate pseudo feedback for reasoning tasks by framing the labeling of solutions to reason problems a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.16345","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.16345/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:14:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"udZdkqGUP3JfKwX4flpJoT8gRMVY/eW8+mbrDDxwE2/VqIIiPnazoa3PBjxPl0faoK1ML1cgtlaOfDF38pV5BA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T08:13:01.648259Z"},"content_sha256":"a8b3b5d902a2457bd63a3d897f70b806ada6c3c5ead1b9c6125e3b63527f519c","schema_version":"1.0","event_id":"sha256:a8b3b5d902a2457bd63a3d897f70b806ada6c3c5ead1b9c6125e3b63527f519c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/SAASNGDQHDTBCST3UVG6DOMNUK/bundle.json","state_url":"https://pith.science/pith/SAASNGDQHDTBCST3UVG6DOMNUK/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/SAASNGDQHDTBCST3UVG6DOMNUK/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T08:13:01Z","links":{"resolver":"https://pith.science/pith/SAASNGDQHDTBCST3UVG6DOMNUK","bundle":"https://pith.science/pith/SAASNGDQHDTBCST3UVG6DOMNUK/bundle.json","state":"https://pith.science/pith/SAASNGDQHDTBCST3UVG6DOMNUK/state.json","well_known_bundle":"https://pith.science/.well-known/pith/SAASNGDQHDTBCST3UVG6DOMNUK/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:SAASNGDQHDTBCST3UVG6DOMNUK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c0574412136553627d35b95e67c68cdec371a566d4b66d14d248034047ca720e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-25T12:44:02Z","title_canon_sha256":"81595d98840b2bd133d9d6fdab122236bd1baa0c37f03d7f42641a5b5d765081"},"schema_version":"1.0","source":{"id":"2411.16345","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2411.16345","created_at":"2026-07-05T10:14:07Z"},{"alias_kind":"arxiv_version","alias_value":"2411.16345v2","created_at":"2026-07-05T10:14:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.16345","created_at":"2026-07-05T10:14:07Z"},{"alias_kind":"pith_short_12","alias_value":"SAASNGDQHDTB","created_at":"2026-07-05T10:14:07Z"},{"alias_kind":"pith_short_16","alias_value":"SAASNGDQHDTBCST3","created_at":"2026-07-05T10:14:07Z"},{"alias_kind":"pith_short_8","alias_value":"SAASNGDQ","created_at":"2026-07-05T10:14:07Z"}],"graph_snapshots":[{"event_id":"sha256:a8b3b5d902a2457bd63a3d897f70b806ada6c3c5ead1b9c6125e3b63527f519c","target":"graph","created_at":"2026-07-05T10:14:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2411.16345/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Preference optimization techniques, such as Direct Preference Optimization (DPO), are frequently employed to enhance the reasoning capabilities of large language models (LLMs) in domains like mathematical reasoning and coding, typically following supervised fine-tuning. These methods rely on high-quality labels for reasoning tasks to generate preference pairs; however, the availability of reasoning datasets with human-verified labels is limited. In this study, we introduce a novel approach to generate pseudo feedback for reasoning tasks by framing the labeling of solutions to reason problems a","authors_text":"Fangkai Jiao, Furu Wei, Geyang Guo, Nancy F. Chen, Shafiq Joty, Xingxing Zhang","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-25T12:44:02Z","title":"Preference Optimization for Reasoning with Pseudo Feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.16345","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:a3ee9ab94f49ed13cbac9d8b3ec9dea4b8782209a6b549eb1623c74cf5aa0d2c","target":"record","created_at":"2026-07-05T10:14:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c0574412136553627d35b95e67c68cdec371a566d4b66d14d248034047ca720e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-25T12:44:02Z","title_canon_sha256":"81595d98840b2bd133d9d6fdab122236bd1baa0c37f03d7f42641a5b5d765081"},"schema_version":"1.0","source":{"id":"2411.16345","kind":"arxiv","version":2}},"canonical_sha256":"900126987038e6114a7ba54de1b98da299cb9117a987fa3aada5344f4c50fc69","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"900126987038e6114a7ba54de1b98da299cb9117a987fa3aada5344f4c50fc69","first_computed_at":"2026-07-05T10:14:07.212464Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:14:07.212464Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Wzu+7ZVPhl4PPnvefVzMuSIizdYtQa68K0YG4+jHwsx8NuazSXkyPxMtuf0L4t88xaJA453jXWabgz0yZ1iKDg==","signature_status":"signed_v1","signed_at":"2026-07-05T10:14:07.212952Z","signed_message":"canonical_sha256_bytes"},"source_id":"2411.16345","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:a3ee9ab94f49ed13cbac9d8b3ec9dea4b8782209a6b549eb1623c74cf5aa0d2c","sha256:a8b3b5d902a2457bd63a3d897f70b806ada6c3c5ead1b9c6125e3b63527f519c"],"state_sha256":"a5ff4ae28338a6989b9046bea5594b23e7edd9f10d822e227c5226c143e06b03"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"eLT/FSlHeb/1XNuh/3vsFgpQL/6GA6N8zY5PAvLt16h26Ch74IDZBNk2E8WyqhZqTnTjMXWp5MJN2Xb2dWNLCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T08:13:01.653748Z","bundle_sha256":"40816ad8822d1d47e93a37518ace0da865075f4116bc755da189170e8882dcce"}}