{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RNC25EC6ABFWIZGC4RKQMQBFHK","short_pith_number":"pith:RNC25EC6","schema_version":"1.0","canonical_sha256":"8b45ae905e004b6464c2e4550640253a8a2917f00ecb7fbbe3d0beeedab3322a","source":{"kind":"arxiv","id":"2505.13346","version":3},"attestation_state":"computed","paper":{"title":"J4R: Learning to Judge with Equivalent Initial State Group Relative Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Austin Xu, Caiming Xiong, Shafiq Joty, Xuan-Phi Nguyen, Yilun Zhou","submitted_at":"2025-05-19T16:50:35Z","abstract_excerpt":"To keep pace with the increasing pace of large language models (LLM) development, model output evaluation has transitioned away from time-consuming human evaluation to automatic evaluation, where LLMs themselves are tasked with assessing and critiquing other model outputs. LLM-as-judge models are a class of generative evaluators that excel in evaluating relatively simple domains, like chat quality, but struggle in reasoning intensive domains where model responses contain more substantive and challenging content. To remedy existing judge shortcomings, we explore training judges with reinforceme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.13346","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-19T16:50:35Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ccecd14f68a30d86ab2e51d69880b64da1762145ca662c83820187a97025f324","abstract_canon_sha256":"9d4f366c2218bdae35bb4271891d9dee808ef008ffa271a7313f00df7799f195"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:23:28.205428Z","signature_b64":"l2cwjFPuti/gXeJdXVnkfGdAGfqSAeC28OLhXxhSdNZDhevkrPB1/BUzrdSupUtzI1eWRHUbMoVeUztUoBc6Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b45ae905e004b6464c2e4550640253a8a2917f00ecb7fbbe3d0beeedab3322a","last_reissued_at":"2026-07-05T11:23:28.204913Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:23:28.204913Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"J4R: Learning to Judge with Equivalent Initial State Group Relative Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Austin Xu, Caiming Xiong, Shafiq Joty, Xuan-Phi Nguyen, Yilun Zhou","submitted_at":"2025-05-19T16:50:35Z","abstract_excerpt":"To keep pace with the increasing pace of large language models (LLM) development, model output evaluation has transitioned away from time-consuming human evaluation to automatic evaluation, where LLMs themselves are tasked with assessing and critiquing other model outputs. LLM-as-judge models are a class of generative evaluators that excel in evaluating relatively simple domains, like chat quality, but struggle in reasoning intensive domains where model responses contain more substantive and challenging content. To remedy existing judge shortcomings, we explore training judges with reinforceme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.13346","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.13346/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.13346","created_at":"2026-07-05T11:23:28.204968+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.13346v3","created_at":"2026-07-05T11:23:28.204968+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.13346","created_at":"2026-07-05T11:23:28.204968+00:00"},{"alias_kind":"pith_short_12","alias_value":"RNC25EC6ABFW","created_at":"2026-07-05T11:23:28.204968+00:00"},{"alias_kind":"pith_short_16","alias_value":"RNC25EC6ABFWIZGC","created_at":"2026-07-05T11:23:28.204968+00:00"},{"alias_kind":"pith_short_8","alias_value":"RNC25EC6","created_at":"2026-07-05T11:23:28.204968+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.23542","citing_title":"On the Shelf Life of Fine-Tuned LLM-Judges: Future-Proofing, Backward-Compatibility, and Question Generalization","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RNC25EC6ABFWIZGC4RKQMQBFHK","json":"https://pith.science/pith/RNC25EC6ABFWIZGC4RKQMQBFHK.json","graph_json":"https://pith.science/api/pith-number/RNC25EC6ABFWIZGC4RKQMQBFHK/graph.json","events_json":"https://pith.science/api/pith-number/RNC25EC6ABFWIZGC4RKQMQBFHK/events.json","paper":"https://pith.science/paper/RNC25EC6"},"agent_actions":{"view_html":"https://pith.science/pith/RNC25EC6ABFWIZGC4RKQMQBFHK","download_json":"https://pith.science/pith/RNC25EC6ABFWIZGC4RKQMQBFHK.json","view_paper":"https://pith.science/paper/RNC25EC6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.13346&json=true","fetch_graph":"https://pith.science/api/pith-number/RNC25EC6ABFWIZGC4RKQMQBFHK/graph.json","fetch_events":"https://pith.science/api/pith-number/RNC25EC6ABFWIZGC4RKQMQBFHK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RNC25EC6ABFWIZGC4RKQMQBFHK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RNC25EC6ABFWIZGC4RKQMQBFHK/action/storage_attestation","attest_author":"https://pith.science/pith/RNC25EC6ABFWIZGC4RKQMQBFHK/action/author_attestation","sign_citation":"https://pith.science/pith/RNC25EC6ABFWIZGC4RKQMQBFHK/action/citation_signature","submit_replication":"https://pith.science/pith/RNC25EC6ABFWIZGC4RKQMQBFHK/action/replication_record"}},"created_at":"2026-07-05T11:23:28.204968+00:00","updated_at":"2026-07-05T11:23:28.204968+00:00"}