{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:6PH2ZX2USXOM5F7IEUY73LNXUK","short_pith_number":"pith:6PH2ZX2U","canonical_record":{"source":{"id":"2410.01729","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T16:39:58Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"0e510d353831e6e60e4b2c22acd75fb35b07d47b43ca7dac242a20957089feb8","abstract_canon_sha256":"3ad7d89d045aa3ad100fc2ec857686113a56a74890a2aa64b4249c7f0f622fa8"},"schema_version":"1.0"},"canonical_sha256":"f3cfacdf5495dcce97e82531fdadb7a2a38c3edc68abc2dd98bfd52f35dd69c4","source":{"kind":"arxiv","id":"2410.01729","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.01729","created_at":"2026-07-05T09:14:51Z"},{"alias_kind":"arxiv_version","alias_value":"2410.01729v1","created_at":"2026-07-05T09:14:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01729","created_at":"2026-07-05T09:14:51Z"},{"alias_kind":"pith_short_12","alias_value":"6PH2ZX2USXOM","created_at":"2026-07-05T09:14:51Z"},{"alias_kind":"pith_short_16","alias_value":"6PH2ZX2USXOM5F7I","created_at":"2026-07-05T09:14:51Z"},{"alias_kind":"pith_short_8","alias_value":"6PH2ZX2U","created_at":"2026-07-05T09:14:51Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:6PH2ZX2USXOM5F7IEUY73LNXUK","target":"record","payload":{"canonical_record":{"source":{"id":"2410.01729","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T16:39:58Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"0e510d353831e6e60e4b2c22acd75fb35b07d47b43ca7dac242a20957089feb8","abstract_canon_sha256":"3ad7d89d045aa3ad100fc2ec857686113a56a74890a2aa64b4249c7f0f622fa8"},"schema_version":"1.0"},"canonical_sha256":"f3cfacdf5495dcce97e82531fdadb7a2a38c3edc68abc2dd98bfd52f35dd69c4","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:51.296463Z","signature_b64":"rB+J+QlJRnvD7koJMrF3j33l0c7UTKCiNfCzyoz7Z6vbS02npP14RsRi7yhWjkBxe/ujcVAq25bxVpOqE05LCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3cfacdf5495dcce97e82531fdadb7a2a38c3edc68abc2dd98bfd52f35dd69c4","last_reissued_at":"2026-07-05T09:14:51.295970Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:51.295970Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2410.01729","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:14:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IfEJny47ff81C32BHSefXamhAgHbYhLcJCPpAHcAowcZFPJNxdmEana81cZBhsA/0TufeA/a+sLKsRy/3mKEBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T06:11:50.748637Z"},"content_sha256":"fcdd8b729e353a0ad5d935aba780518b664a6fbbc36890a7dd581df4e6a284fc","schema_version":"1.0","event_id":"sha256:fcdd8b729e353a0ad5d935aba780518b664a6fbbc36890a7dd581df4e6a284fc"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:6PH2ZX2USXOM5F7IEUY73LNXUK","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Evaluating Robustness of Reward Models for Mathematical Reasoning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Dongha Lee, Dongjin Kang, Hyungjoo Chae, Jinyoung Yeo, Jungsoo Won, Sunghwan Kim, Taeyoon Kwon","submitted_at":"2024-10-02T16:39:58Z","abstract_excerpt":"Reward models are key in reinforcement learning from human feedback (RLHF) systems, aligning the model behavior with human preferences. Particularly in the math domain, there have been plenty of studies using reward models to align policies for improving reasoning capabilities. Recently, as the importance of reward models has been emphasized, RewardBench is proposed to understand their behavior. However, we figure out that the math subset of RewardBench has different representations between chosen and rejected completions, and relies on a single comparison, which may lead to unreliable results"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01729","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01729/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:14:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Yma//DpZNswcIFB6/2yGI+xWMgTd7pw/Jl86Hwgr5Qlw/JGrslGZdmfe/Qv4lwI+oY7uTsuBNAr2UVFoVZ6JAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T06:11:50.749134Z"},"content_sha256":"c49390ba9551b31378f921c90eae8cb93bb41d4b8bec9af45e068fa0f8eb8acf","schema_version":"1.0","event_id":"sha256:c49390ba9551b31378f921c90eae8cb93bb41d4b8bec9af45e068fa0f8eb8acf"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/6PH2ZX2USXOM5F7IEUY73LNXUK/bundle.json","state_url":"https://pith.science/pith/6PH2ZX2USXOM5F7IEUY73LNXUK/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/6PH2ZX2USXOM5F7IEUY73LNXUK/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T06:11:50Z","links":{"resolver":"https://pith.science/pith/6PH2ZX2USXOM5F7IEUY73LNXUK","bundle":"https://pith.science/pith/6PH2ZX2USXOM5F7IEUY73LNXUK/bundle.json","state":"https://pith.science/pith/6PH2ZX2USXOM5F7IEUY73LNXUK/state.json","well_known_bundle":"https://pith.science/.well-known/pith/6PH2ZX2USXOM5F7IEUY73LNXUK/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:6PH2ZX2USXOM5F7IEUY73LNXUK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"3ad7d89d045aa3ad100fc2ec857686113a56a74890a2aa64b4249c7f0f622fa8","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T16:39:58Z","title_canon_sha256":"0e510d353831e6e60e4b2c22acd75fb35b07d47b43ca7dac242a20957089feb8"},"schema_version":"1.0","source":{"id":"2410.01729","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.01729","created_at":"2026-07-05T09:14:51Z"},{"alias_kind":"arxiv_version","alias_value":"2410.01729v1","created_at":"2026-07-05T09:14:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01729","created_at":"2026-07-05T09:14:51Z"},{"alias_kind":"pith_short_12","alias_value":"6PH2ZX2USXOM","created_at":"2026-07-05T09:14:51Z"},{"alias_kind":"pith_short_16","alias_value":"6PH2ZX2USXOM5F7I","created_at":"2026-07-05T09:14:51Z"},{"alias_kind":"pith_short_8","alias_value":"6PH2ZX2U","created_at":"2026-07-05T09:14:51Z"}],"graph_snapshots":[{"event_id":"sha256:c49390ba9551b31378f921c90eae8cb93bb41d4b8bec9af45e068fa0f8eb8acf","target":"graph","created_at":"2026-07-05T09:14:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.01729/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reward models are key in reinforcement learning from human feedback (RLHF) systems, aligning the model behavior with human preferences. Particularly in the math domain, there have been plenty of studies using reward models to align policies for improving reasoning capabilities. Recently, as the importance of reward models has been emphasized, RewardBench is proposed to understand their behavior. However, we figure out that the math subset of RewardBench has different representations between chosen and rejected completions, and relies on a single comparison, which may lead to unreliable results","authors_text":"Dongha Lee, Dongjin Kang, Hyungjoo Chae, Jinyoung Yeo, Jungsoo Won, Sunghwan Kim, Taeyoon Kwon","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T16:39:58Z","title":"Evaluating Robustness of Reward Models for Mathematical Reasoning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01729","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:fcdd8b729e353a0ad5d935aba780518b664a6fbbc36890a7dd581df4e6a284fc","target":"record","created_at":"2026-07-05T09:14:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"3ad7d89d045aa3ad100fc2ec857686113a56a74890a2aa64b4249c7f0f622fa8","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T16:39:58Z","title_canon_sha256":"0e510d353831e6e60e4b2c22acd75fb35b07d47b43ca7dac242a20957089feb8"},"schema_version":"1.0","source":{"id":"2410.01729","kind":"arxiv","version":1}},"canonical_sha256":"f3cfacdf5495dcce97e82531fdadb7a2a38c3edc68abc2dd98bfd52f35dd69c4","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f3cfacdf5495dcce97e82531fdadb7a2a38c3edc68abc2dd98bfd52f35dd69c4","first_computed_at":"2026-07-05T09:14:51.295970Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:14:51.295970Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"rB+J+QlJRnvD7koJMrF3j33l0c7UTKCiNfCzyoz7Z6vbS02npP14RsRi7yhWjkBxe/ujcVAq25bxVpOqE05LCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T09:14:51.296463Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.01729","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:fcdd8b729e353a0ad5d935aba780518b664a6fbbc36890a7dd581df4e6a284fc","sha256:c49390ba9551b31378f921c90eae8cb93bb41d4b8bec9af45e068fa0f8eb8acf"],"state_sha256":"6742d2d794c624c3119eda269116bda39cc244edbc6387c8905d509489270c9e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UxzVli0xpKaWkUi5XB80oVSgue4Zg6+g7Q+7YEBk6S3jjlUCqX9Z1PKo5CVyNpFGHHOfHVaz8q6uD2Gmsk9OCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T06:11:50.752899Z","bundle_sha256":"bb84352f34ffc674918ea62bd8e7761f63a864b07dbaf8e0ddd97a5f0a9ef848"}}