{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:ZGUIERMNKUHFVSDWJV2QPY2I2Z","short_pith_number":"pith:ZGUIERMN","schema_version":"1.0","canonical_sha256":"c9a882458d550e5ac8764d7507e348d642b104fc13081161fb45df80a60ac3ea","source":{"kind":"arxiv","id":"2602.17658","version":2},"attestation_state":"computed","paper":{"title":"MARS: Margin and Semantic-Aware Data Augmentation for Reward Modeling","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Osvaldo Simeone, Payel Bhattacharjee, Ravi Tandon","submitted_at":"2026-02-19T18:59:03Z","abstract_excerpt":"Reward modeling is central to alignment pipelines such as RLHF, RLAIF, and PPO-based policy optimization, yet its reliability is constrained by limited and heterogeneous human preference data that are expensive to collect at scale. While synthetic augmentation can expand preference supervision, existing methods often augment uniformly or at the representation level, without targeting examples where the reward model is uncertain or prone to mis-ranking. In this paper, we introduce MARS (Margin and Semantic-Aware Data Augmentation for Reward Modeling), an adaptive augmentation framework that pri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.17658","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2026-02-19T18:59:03Z","cross_cats_sorted":["cs.AI","cs.IT","math.IT"],"title_canon_sha256":"609da6b8d1d026d796e8c5e21f105a5d390cf4c46d18a342f021af41ce833d8d","abstract_canon_sha256":"e438fa90d40b4e6901e2d3afd80d462066af12ef07d44315346ae53a1df97267"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-26T02:04:06.757998Z","signature_b64":"hg073ECpIashKF0d7Kl3Zqy7TKEZOjEHXlk4adypeirqJr5j7sfRLpyJ7ZjzSJAfUo7wb6dKMK3qkxtZ+Ne9BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9a882458d550e5ac8764d7507e348d642b104fc13081161fb45df80a60ac3ea","last_reissued_at":"2026-05-26T02:04:06.757162Z","signature_status":"signed_v1","first_computed_at":"2026-05-26T02:04:06.757162Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MARS: Margin and Semantic-Aware Data Augmentation for Reward Modeling","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Osvaldo Simeone, Payel Bhattacharjee, Ravi Tandon","submitted_at":"2026-02-19T18:59:03Z","abstract_excerpt":"Reward modeling is central to alignment pipelines such as RLHF, RLAIF, and PPO-based policy optimization, yet its reliability is constrained by limited and heterogeneous human preference data that are expensive to collect at scale. While synthetic augmentation can expand preference supervision, existing methods often augment uniformly or at the representation level, without targeting examples where the reward model is uncertain or prone to mis-ranking. In this paper, we introduce MARS (Margin and Semantic-Aware Data Augmentation for Reward Modeling), an adaptive augmentation framework that pri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.17658","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.17658/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.17658","created_at":"2026-05-26T02:04:06.757273+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.17658v2","created_at":"2026-05-26T02:04:06.757273+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.17658","created_at":"2026-05-26T02:04:06.757273+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZGUIERMNKUHF","created_at":"2026-05-26T02:04:06.757273+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZGUIERMNKUHFVSDW","created_at":"2026-05-26T02:04:06.757273+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZGUIERMN","created_at":"2026-05-26T02:04:06.757273+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZGUIERMNKUHFVSDWJV2QPY2I2Z","json":"https://pith.science/pith/ZGUIERMNKUHFVSDWJV2QPY2I2Z.json","graph_json":"https://pith.science/api/pith-number/ZGUIERMNKUHFVSDWJV2QPY2I2Z/graph.json","events_json":"https://pith.science/api/pith-number/ZGUIERMNKUHFVSDWJV2QPY2I2Z/events.json","paper":"https://pith.science/paper/ZGUIERMN"},"agent_actions":{"view_html":"https://pith.science/pith/ZGUIERMNKUHFVSDWJV2QPY2I2Z","download_json":"https://pith.science/pith/ZGUIERMNKUHFVSDWJV2QPY2I2Z.json","view_paper":"https://pith.science/paper/ZGUIERMN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.17658&json=true","fetch_graph":"https://pith.science/api/pith-number/ZGUIERMNKUHFVSDWJV2QPY2I2Z/graph.json","fetch_events":"https://pith.science/api/pith-number/ZGUIERMNKUHFVSDWJV2QPY2I2Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZGUIERMNKUHFVSDWJV2QPY2I2Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZGUIERMNKUHFVSDWJV2QPY2I2Z/action/storage_attestation","attest_author":"https://pith.science/pith/ZGUIERMNKUHFVSDWJV2QPY2I2Z/action/author_attestation","sign_citation":"https://pith.science/pith/ZGUIERMNKUHFVSDWJV2QPY2I2Z/action/citation_signature","submit_replication":"https://pith.science/pith/ZGUIERMNKUHFVSDWJV2QPY2I2Z/action/replication_record"}},"created_at":"2026-05-26T02:04:06.757273+00:00","updated_at":"2026-05-26T02:04:06.757273+00:00"}