{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:DNN7WYXLVYWKUKOKSK4HOSD6DL","short_pith_number":"pith:DNN7WYXL","canonical_record":{"source":{"id":"2502.16182","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-22T10:59:11Z","cross_cats_sorted":[],"title_canon_sha256":"4c99e0e02255d15148fe541c76578b5cb91c4b93de302d9aa2bb2b743e8efdd8","abstract_canon_sha256":"b63fe45f005a500be2ac7c8ae09a95a508865369cc51fc819d17c9953ba86d63"},"schema_version":"1.0"},"canonical_sha256":"1b5bfb62ebae2caa29ca92b877487e1ae6dcd44986ecc4c394c3e1226d5cd6b8","source":{"kind":"arxiv","id":"2502.16182","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.16182","created_at":"2026-07-05T10:35:46Z"},{"alias_kind":"arxiv_version","alias_value":"2502.16182v2","created_at":"2026-07-05T10:35:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.16182","created_at":"2026-07-05T10:35:46Z"},{"alias_kind":"pith_short_12","alias_value":"DNN7WYXLVYWK","created_at":"2026-07-05T10:35:46Z"},{"alias_kind":"pith_short_16","alias_value":"DNN7WYXLVYWKUKOK","created_at":"2026-07-05T10:35:46Z"},{"alias_kind":"pith_short_8","alias_value":"DNN7WYXL","created_at":"2026-07-05T10:35:46Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:DNN7WYXLVYWKUKOKSK4HOSD6DL","target":"record","payload":{"canonical_record":{"source":{"id":"2502.16182","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-22T10:59:11Z","cross_cats_sorted":[],"title_canon_sha256":"4c99e0e02255d15148fe541c76578b5cb91c4b93de302d9aa2bb2b743e8efdd8","abstract_canon_sha256":"b63fe45f005a500be2ac7c8ae09a95a508865369cc51fc819d17c9953ba86d63"},"schema_version":"1.0"},"canonical_sha256":"1b5bfb62ebae2caa29ca92b877487e1ae6dcd44986ecc4c394c3e1226d5cd6b8","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:35:46.664593Z","signature_b64":"DcI7QKNT2Rqsa4CSggil/8FzcEdrQdPx+GxtFSvFXoPojafy6SVHr18k7rEFerZT5t4QrWFhkM5P4E6aedKJCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b5bfb62ebae2caa29ca92b877487e1ae6dcd44986ecc4c394c3e1226d5cd6b8","last_reissued_at":"2026-07-05T10:35:46.663728Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:35:46.663728Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2502.16182","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:35:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2zbGqgtgMh7i/FKRUASQlhAkqDssJ0Y0fkJlnua9vhoHKetE1TNL5LNE6BtH21dqs+8SMR+9m3hd0MnunK52Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T13:04:07.461887Z"},"content_sha256":"1cfd4d316de3ec9f130751aac7971229f028890976fe32d00731daea56571793","schema_version":"1.0","event_id":"sha256:1cfd4d316de3ec9f130751aac7971229f028890976fe32d00731daea56571793"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:DNN7WYXLVYWKUKOKSK4HOSD6DL","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"IPO: Your Language Model is Secretly a Preference Classifier","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ayush Singh, Paras Chopra, Shivank Garg, Shweta Singh","submitted_at":"2025-02-22T10:59:11Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has emerged as the primary method for aligning large language models (LLMs) with human preferences. While it enables LLMs to achieve human-level alignment, it often incurs significant computational and financial costs due to its reliance on training external reward models or human-labeled preferences. In this work, we propose Implicit Preference Optimization (IPO), an alternative approach that leverages generative LLMs as preference classifiers, thereby reducing the dependence on external human feedback or reward models to obtain preferences. W"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.16182","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.16182/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:35:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"7+wjuOEyAdFh1T6k4WHXZ16wtXdlqfoJ0dJn4J8nQFHTl0qP0a7ZnSzS0RnGv2iVyxrNUBcCYtGIy35ZdSCrBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T13:04:07.462454Z"},"content_sha256":"9b9d2420d4ee362a7cd4682391ee5cd19cf3b9519d3f5a325db74160b853aa4c","schema_version":"1.0","event_id":"sha256:9b9d2420d4ee362a7cd4682391ee5cd19cf3b9519d3f5a325db74160b853aa4c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/DNN7WYXLVYWKUKOKSK4HOSD6DL/bundle.json","state_url":"https://pith.science/pith/DNN7WYXLVYWKUKOKSK4HOSD6DL/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/DNN7WYXLVYWKUKOKSK4HOSD6DL/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-19T13:04:07Z","links":{"resolver":"https://pith.science/pith/DNN7WYXLVYWKUKOKSK4HOSD6DL","bundle":"https://pith.science/pith/DNN7WYXLVYWKUKOKSK4HOSD6DL/bundle.json","state":"https://pith.science/pith/DNN7WYXLVYWKUKOKSK4HOSD6DL/state.json","well_known_bundle":"https://pith.science/.well-known/pith/DNN7WYXLVYWKUKOKSK4HOSD6DL/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:DNN7WYXLVYWKUKOKSK4HOSD6DL","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b63fe45f005a500be2ac7c8ae09a95a508865369cc51fc819d17c9953ba86d63","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-22T10:59:11Z","title_canon_sha256":"4c99e0e02255d15148fe541c76578b5cb91c4b93de302d9aa2bb2b743e8efdd8"},"schema_version":"1.0","source":{"id":"2502.16182","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.16182","created_at":"2026-07-05T10:35:46Z"},{"alias_kind":"arxiv_version","alias_value":"2502.16182v2","created_at":"2026-07-05T10:35:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.16182","created_at":"2026-07-05T10:35:46Z"},{"alias_kind":"pith_short_12","alias_value":"DNN7WYXLVYWK","created_at":"2026-07-05T10:35:46Z"},{"alias_kind":"pith_short_16","alias_value":"DNN7WYXLVYWKUKOK","created_at":"2026-07-05T10:35:46Z"},{"alias_kind":"pith_short_8","alias_value":"DNN7WYXL","created_at":"2026-07-05T10:35:46Z"}],"graph_snapshots":[{"event_id":"sha256:9b9d2420d4ee362a7cd4682391ee5cd19cf3b9519d3f5a325db74160b853aa4c","target":"graph","created_at":"2026-07-05T10:35:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.16182/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has emerged as the primary method for aligning large language models (LLMs) with human preferences. While it enables LLMs to achieve human-level alignment, it often incurs significant computational and financial costs due to its reliance on training external reward models or human-labeled preferences. In this work, we propose Implicit Preference Optimization (IPO), an alternative approach that leverages generative LLMs as preference classifiers, thereby reducing the dependence on external human feedback or reward models to obtain preferences. W","authors_text":"Ayush Singh, Paras Chopra, Shivank Garg, Shweta Singh","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-22T10:59:11Z","title":"IPO: Your Language Model is Secretly a Preference Classifier"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.16182","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1cfd4d316de3ec9f130751aac7971229f028890976fe32d00731daea56571793","target":"record","created_at":"2026-07-05T10:35:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b63fe45f005a500be2ac7c8ae09a95a508865369cc51fc819d17c9953ba86d63","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-22T10:59:11Z","title_canon_sha256":"4c99e0e02255d15148fe541c76578b5cb91c4b93de302d9aa2bb2b743e8efdd8"},"schema_version":"1.0","source":{"id":"2502.16182","kind":"arxiv","version":2}},"canonical_sha256":"1b5bfb62ebae2caa29ca92b877487e1ae6dcd44986ecc4c394c3e1226d5cd6b8","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"1b5bfb62ebae2caa29ca92b877487e1ae6dcd44986ecc4c394c3e1226d5cd6b8","first_computed_at":"2026-07-05T10:35:46.663728Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:35:46.663728Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"DcI7QKNT2Rqsa4CSggil/8FzcEdrQdPx+GxtFSvFXoPojafy6SVHr18k7rEFerZT5t4QrWFhkM5P4E6aedKJCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T10:35:46.664593Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.16182","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1cfd4d316de3ec9f130751aac7971229f028890976fe32d00731daea56571793","sha256:9b9d2420d4ee362a7cd4682391ee5cd19cf3b9519d3f5a325db74160b853aa4c"],"state_sha256":"a3831ab86e9ffe886d345e07acb454c1d6f4b0d4325d8def74b8473e3e42f9c5"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"mQH/50QscS4fRGhkOhhSeSbTv2bdrQX76X5tOXVv5s9dv6QnzcThqCpIZ2g/qHmsQxqNberR+/WzHuvb7hiFDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-19T13:04:07.468010Z","bundle_sha256":"6524118c0147006d51f5b35d6ecb695f10851dbe859f762d3b8646f300a5b728"}}