{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3LFJNDUY5HXMIQ4AFVISI2N6FM","short_pith_number":"pith:3LFJNDUY","schema_version":"1.0","canonical_sha256":"daca968e98e9eec443802d512469be2b34aaca614b0ce3c66f10b6b2fdc355d8","source":{"kind":"arxiv","id":"2410.05495","version":1},"attestation_state":"computed","paper":{"title":"Self-rationalization improves LLM as a fine-grained judge","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aditya Gulati, Jahnavi Jambholkar, James Zou, Keith Stevens, Meghana Arakkal Rajeev, Nazneen Rajani, Oliver Molenschot, Prapti Trivedi, Rajkumar Ramamurthy, Tanveesh Singh Chaudhery","submitted_at":"2024-10-07T21:05:53Z","abstract_excerpt":"LLM-as-a-judge models have been used for evaluating both human and AI generated content, specifically by providing scores and rationales. Rationales, in addition to increasing transparency, help models learn to calibrate its judgments. Enhancing a model's rationale can therefore improve its calibration abilities and ultimately the ability to score content. We introduce Self-Rationalization, an iterative process of improving the rationales for the judge models, which consequently improves the score for fine-grained customizable scoring criteria (i.e., likert-scale scoring with arbitrary evaluat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05495","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-07T21:05:53Z","cross_cats_sorted":[],"title_canon_sha256":"5686cca1e51376af8136dd49c757e5981740d14bcfcc62f740e0eab94ac50a7c","abstract_canon_sha256":"7e2ce6ecb082c71f34470d1d919f66e7e5255dc99fe972017c3205fa563e3539"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:20.187833Z","signature_b64":"mzX5ZWve2DqT9L+Dp/SODeVvFP1OgL73DqZWvsCnvB7Kh9eztoOslG+iLNSSjv4vYmDvBpo9Qo22PW8OglJ6Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"daca968e98e9eec443802d512469be2b34aaca614b0ce3c66f10b6b2fdc355d8","last_reissued_at":"2026-07-05T09:17:20.187358Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:20.187358Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-rationalization improves LLM as a fine-grained judge","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aditya Gulati, Jahnavi Jambholkar, James Zou, Keith Stevens, Meghana Arakkal Rajeev, Nazneen Rajani, Oliver Molenschot, Prapti Trivedi, Rajkumar Ramamurthy, Tanveesh Singh Chaudhery","submitted_at":"2024-10-07T21:05:53Z","abstract_excerpt":"LLM-as-a-judge models have been used for evaluating both human and AI generated content, specifically by providing scores and rationales. Rationales, in addition to increasing transparency, help models learn to calibrate its judgments. Enhancing a model's rationale can therefore improve its calibration abilities and ultimately the ability to score content. We introduce Self-Rationalization, an iterative process of improving the rationales for the judge models, which consequently improves the score for fine-grained customizable scoring criteria (i.e., likert-scale scoring with arbitrary evaluat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05495","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05495/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05495","created_at":"2026-07-05T09:17:20.187414+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05495v1","created_at":"2026-07-05T09:17:20.187414+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05495","created_at":"2026-07-05T09:17:20.187414+00:00"},{"alias_kind":"pith_short_12","alias_value":"3LFJNDUY5HXM","created_at":"2026-07-05T09:17:20.187414+00:00"},{"alias_kind":"pith_short_16","alias_value":"3LFJNDUY5HXMIQ4A","created_at":"2026-07-05T09:17:20.187414+00:00"},{"alias_kind":"pith_short_8","alias_value":"3LFJNDUY","created_at":"2026-07-05T09:17:20.187414+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":228,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15676","citing_title":"EvoRAG: Making Knowledge Graph-based RAG Automatically Evolve through Feedback-driven Backpropagation","ref_index":82,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3LFJNDUY5HXMIQ4AFVISI2N6FM","json":"https://pith.science/pith/3LFJNDUY5HXMIQ4AFVISI2N6FM.json","graph_json":"https://pith.science/api/pith-number/3LFJNDUY5HXMIQ4AFVISI2N6FM/graph.json","events_json":"https://pith.science/api/pith-number/3LFJNDUY5HXMIQ4AFVISI2N6FM/events.json","paper":"https://pith.science/paper/3LFJNDUY"},"agent_actions":{"view_html":"https://pith.science/pith/3LFJNDUY5HXMIQ4AFVISI2N6FM","download_json":"https://pith.science/pith/3LFJNDUY5HXMIQ4AFVISI2N6FM.json","view_paper":"https://pith.science/paper/3LFJNDUY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05495&json=true","fetch_graph":"https://pith.science/api/pith-number/3LFJNDUY5HXMIQ4AFVISI2N6FM/graph.json","fetch_events":"https://pith.science/api/pith-number/3LFJNDUY5HXMIQ4AFVISI2N6FM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3LFJNDUY5HXMIQ4AFVISI2N6FM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3LFJNDUY5HXMIQ4AFVISI2N6FM/action/storage_attestation","attest_author":"https://pith.science/pith/3LFJNDUY5HXMIQ4AFVISI2N6FM/action/author_attestation","sign_citation":"https://pith.science/pith/3LFJNDUY5HXMIQ4AFVISI2N6FM/action/citation_signature","submit_replication":"https://pith.science/pith/3LFJNDUY5HXMIQ4AFVISI2N6FM/action/replication_record"}},"created_at":"2026-07-05T09:17:20.187414+00:00","updated_at":"2026-07-05T09:17:20.187414+00:00"}