{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:3AFIGSWKSMPF5PM2LWMDPAWP5U","short_pith_number":"pith:3AFIGSWK","canonical_record":{"source":{"id":"2607.03248","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-03T12:04:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b8061188619ef1ea8699a799de470209796d5b7ae03abee300fef3658e9dff31","abstract_canon_sha256":"39b88875881dfba53c68ae8a44a888269b391ef77194e2fcc41a8ea972ca9d01"},"schema_version":"1.0"},"canonical_sha256":"d80a834aca931e5ebd9a5d983782cfed3e21c5453c7b9d8f4f56fada9f6be692","source":{"kind":"arxiv","id":"2607.03248","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.03248","created_at":"2026-07-07T01:16:47Z"},{"alias_kind":"arxiv_version","alias_value":"2607.03248v1","created_at":"2026-07-07T01:16:47Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.03248","created_at":"2026-07-07T01:16:47Z"},{"alias_kind":"pith_short_12","alias_value":"3AFIGSWKSMPF","created_at":"2026-07-07T01:16:47Z"},{"alias_kind":"pith_short_16","alias_value":"3AFIGSWKSMPF5PM2","created_at":"2026-07-07T01:16:47Z"},{"alias_kind":"pith_short_8","alias_value":"3AFIGSWK","created_at":"2026-07-07T01:16:47Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:3AFIGSWKSMPF5PM2LWMDPAWP5U","target":"record","payload":{"canonical_record":{"source":{"id":"2607.03248","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-03T12:04:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b8061188619ef1ea8699a799de470209796d5b7ae03abee300fef3658e9dff31","abstract_canon_sha256":"39b88875881dfba53c68ae8a44a888269b391ef77194e2fcc41a8ea972ca9d01"},"schema_version":"1.0"},"canonical_sha256":"d80a834aca931e5ebd9a5d983782cfed3e21c5453c7b9d8f4f56fada9f6be692","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T01:16:47.841082Z","signature_b64":"0ZdssTVo3REyKO0KXWmBsaGuDEqXmPABbnXY9db/w/NLmwHCvChE5YTVC8YKK3Zd3PwUgfgkeuJUOkSeae49Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d80a834aca931e5ebd9a5d983782cfed3e21c5453c7b9d8f4f56fada9f6be692","last_reissued_at":"2026-07-07T01:16:47.840659Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T01:16:47.840659Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.03248","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-07T01:16:47Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"7y87foqnf00A+yZa8gPqujPjWfdRWET2JPs1sBEA1XXGm6kjGbuGL4KK12dhEtA4Uj0xpMh3VM6jw5fmTJxqDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T02:05:25.366417Z"},"content_sha256":"0cfa886dbb3326ac17ff0aa5c4daa995dbf009c4671528c6f7cbabff5fd89946","schema_version":"1.0","event_id":"sha256:0cfa886dbb3326ac17ff0aa5c4daa995dbf009c4671528c6f7cbabff5fd89946"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:3AFIGSWKSMPF5PM2LWMDPAWP5U","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Unbiased Alignment for Large Language Models with Noisy Preferences","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Haoliang Li, Hui Liu, Jialiang Wang, Xianming Liu, Xiong Zhou","submitted_at":"2026-07-03T12:04:43Z","abstract_excerpt":"The alignment of large language models with human preferences is commonly achieved through Reinforcement Learning from Human Feedback or Direct Preference Optimization. However, these methods are vulnerable to the significant noise prevalent in real-world preference datasets. To address this critical issue, we present a theoretical framework for unbiased alignment, introducing the Unbiased Reward Model (URM) loss and the Unbiased Direct Preference Optimization (UDPO) loss. By mathematically correcting the distortion induced by preference noise, our novel objectives enable unbiased model traini"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.03248","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.03248/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-07T01:16:47Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"3i1S/BHvEZuIYG06aQeLR/TVPkg2S3XH9DIgX1q0ru2TkHbCaoeo538gRfCqqOnaWNSiFqh88Zsd9qJ8mzbPBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T02:05:25.367264Z"},"content_sha256":"8c66934acb6ed2e61fe4fe993c775030696da013985f0e136f52a67563eb08fa","schema_version":"1.0","event_id":"sha256:8c66934acb6ed2e61fe4fe993c775030696da013985f0e136f52a67563eb08fa"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/3AFIGSWKSMPF5PM2LWMDPAWP5U/bundle.json","state_url":"https://pith.science/pith/3AFIGSWKSMPF5PM2LWMDPAWP5U/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/3AFIGSWKSMPF5PM2LWMDPAWP5U/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T02:05:25Z","links":{"resolver":"https://pith.science/pith/3AFIGSWKSMPF5PM2LWMDPAWP5U","bundle":"https://pith.science/pith/3AFIGSWKSMPF5PM2LWMDPAWP5U/bundle.json","state":"https://pith.science/pith/3AFIGSWKSMPF5PM2LWMDPAWP5U/state.json","well_known_bundle":"https://pith.science/.well-known/pith/3AFIGSWKSMPF5PM2LWMDPAWP5U/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:3AFIGSWKSMPF5PM2LWMDPAWP5U","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"39b88875881dfba53c68ae8a44a888269b391ef77194e2fcc41a8ea972ca9d01","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-03T12:04:43Z","title_canon_sha256":"b8061188619ef1ea8699a799de470209796d5b7ae03abee300fef3658e9dff31"},"schema_version":"1.0","source":{"id":"2607.03248","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.03248","created_at":"2026-07-07T01:16:47Z"},{"alias_kind":"arxiv_version","alias_value":"2607.03248v1","created_at":"2026-07-07T01:16:47Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.03248","created_at":"2026-07-07T01:16:47Z"},{"alias_kind":"pith_short_12","alias_value":"3AFIGSWKSMPF","created_at":"2026-07-07T01:16:47Z"},{"alias_kind":"pith_short_16","alias_value":"3AFIGSWKSMPF5PM2","created_at":"2026-07-07T01:16:47Z"},{"alias_kind":"pith_short_8","alias_value":"3AFIGSWK","created_at":"2026-07-07T01:16:47Z"}],"graph_snapshots":[{"event_id":"sha256:8c66934acb6ed2e61fe4fe993c775030696da013985f0e136f52a67563eb08fa","target":"graph","created_at":"2026-07-07T01:16:47Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.03248/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The alignment of large language models with human preferences is commonly achieved through Reinforcement Learning from Human Feedback or Direct Preference Optimization. However, these methods are vulnerable to the significant noise prevalent in real-world preference datasets. To address this critical issue, we present a theoretical framework for unbiased alignment, introducing the Unbiased Reward Model (URM) loss and the Unbiased Direct Preference Optimization (UDPO) loss. By mathematically correcting the distortion induced by preference noise, our novel objectives enable unbiased model traini","authors_text":"Haoliang Li, Hui Liu, Jialiang Wang, Xianming Liu, Xiong Zhou","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-03T12:04:43Z","title":"Unbiased Alignment for Large Language Models with Noisy Preferences"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.03248","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:0cfa886dbb3326ac17ff0aa5c4daa995dbf009c4671528c6f7cbabff5fd89946","target":"record","created_at":"2026-07-07T01:16:47Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"39b88875881dfba53c68ae8a44a888269b391ef77194e2fcc41a8ea972ca9d01","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-03T12:04:43Z","title_canon_sha256":"b8061188619ef1ea8699a799de470209796d5b7ae03abee300fef3658e9dff31"},"schema_version":"1.0","source":{"id":"2607.03248","kind":"arxiv","version":1}},"canonical_sha256":"d80a834aca931e5ebd9a5d983782cfed3e21c5453c7b9d8f4f56fada9f6be692","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d80a834aca931e5ebd9a5d983782cfed3e21c5453c7b9d8f4f56fada9f6be692","first_computed_at":"2026-07-07T01:16:47.840659Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-07T01:16:47.840659Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"0ZdssTVo3REyKO0KXWmBsaGuDEqXmPABbnXY9db/w/NLmwHCvChE5YTVC8YKK3Zd3PwUgfgkeuJUOkSeae49Dg==","signature_status":"signed_v1","signed_at":"2026-07-07T01:16:47.841082Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.03248","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:0cfa886dbb3326ac17ff0aa5c4daa995dbf009c4671528c6f7cbabff5fd89946","sha256:8c66934acb6ed2e61fe4fe993c775030696da013985f0e136f52a67563eb08fa"],"state_sha256":"518ddd744ccf7d54b0acba4ea0d743335464dc490b55f964299780332491929a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"6X3Gum+jH86hfwCyFvtInwwnXdXikMOlfU+zHB6vr50HKmOU4XYgcx75w8uOBjlw1zW1Se8OpmfN5jcSFd7LAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T02:05:25.371617Z","bundle_sha256":"c02d14b4abfb3149d491f8d35605debf618386ab8d1ed3574e3a04607458e375"}}