{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:66R6IVCCSKXHQD3DCPQPRJOGZB","short_pith_number":"pith:66R6IVCC","schema_version":"1.0","canonical_sha256":"f7a3e4544292ae780f6313e0f8a5c6c86aea49dd4af56c1d2fefd53c1de51983","source":{"kind":"arxiv","id":"2501.18099","version":2},"attestation_state":"computed","paper":{"title":"Learning to Plan & Reason for Evaluation with Thinking-LLM-as-a-Judge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Jason Weston, Marjan Ghazvininejad, Swarnadeep Saha, Tianlu Wang, Xian Li","submitted_at":"2025-01-30T02:21:59Z","abstract_excerpt":"LLM-as-a-Judge models generate chain-of-thought (CoT) sequences intended to capture the step-bystep reasoning process that underlies the final evaluation of a response. However, due to the lack of human annotated CoTs for evaluation, the required components and structure of effective reasoning traces remain understudied. Consequently, previous approaches often (1) constrain reasoning traces to hand-designed components, such as a list of criteria, reference answers, or verification questions and (2) structure them such that planning is intertwined with the reasoning for evaluation. In this work"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.18099","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-01-30T02:21:59Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"19d2ab198c68c19253ce68f2e4781be1f8a7b92a927f5de9d5da6bf503773c75","abstract_canon_sha256":"02e9031e7176471ccfea5f3edb489c5f4d8c379e7cb60ba66fa28bc445ddd23e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:33:19.575147Z","signature_b64":"lNyFmKNu3Ax/BVTv4ikxJfI1JcVWIh+FOXd02sDbvT74PmT/i5x4Zb20LBs425lBjmxi/fZeqDfn+Qvf9zEFDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f7a3e4544292ae780f6313e0f8a5c6c86aea49dd4af56c1d2fefd53c1de51983","last_reissued_at":"2026-07-05T11:33:19.574722Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:33:19.574722Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Plan & Reason for Evaluation with Thinking-LLM-as-a-Judge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Jason Weston, Marjan Ghazvininejad, Swarnadeep Saha, Tianlu Wang, Xian Li","submitted_at":"2025-01-30T02:21:59Z","abstract_excerpt":"LLM-as-a-Judge models generate chain-of-thought (CoT) sequences intended to capture the step-bystep reasoning process that underlies the final evaluation of a response. However, due to the lack of human annotated CoTs for evaluation, the required components and structure of effective reasoning traces remain understudied. Consequently, previous approaches often (1) constrain reasoning traces to hand-designed components, such as a list of criteria, reference answers, or verification questions and (2) structure them such that planning is intertwined with the reasoning for evaluation. In this work"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.18099","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.18099/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.18099","created_at":"2026-07-05T11:33:19.574787+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.18099v2","created_at":"2026-07-05T11:33:19.574787+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.18099","created_at":"2026-07-05T11:33:19.574787+00:00"},{"alias_kind":"pith_short_12","alias_value":"66R6IVCCSKXH","created_at":"2026-07-05T11:33:19.574787+00:00"},{"alias_kind":"pith_short_16","alias_value":"66R6IVCCSKXHQD3D","created_at":"2026-07-05T11:33:19.574787+00:00"},{"alias_kind":"pith_short_8","alias_value":"66R6IVCC","created_at":"2026-07-05T11:33:19.574787+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04599","citing_title":"Plan First, Judge Later, Run Better: A DMAIC-Inspired Agentic System for Industrial Anomaly Detection","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":165,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03013","citing_title":"LLMs, You Can Evaluate It! Design of Multi-perspective Report Evaluation for Security Operation Centers","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11299","citing_title":"Primal Generation, Dual Judgment: Self-Training from Test-Time Scaling","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11299","citing_title":"Primal Generation, Dual Judgment: Self-Training from Test-Time Scaling","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10805","citing_title":"Reasoning Is Not Free: Robust Adaptive Cost-Efficient Routing for LLM-as-a-Judge","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07484","citing_title":"ConsistRM: Improving Generative Reward Models via Consistency-Aware Self-Training","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05371","citing_title":"LLM-as-Judge for Semantic Judging of Powerline Segmentation in UAV Inspection","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13371","citing_title":"Empirical Evidence of Complexity-Induced Limits in Large Language Models on Finite Discrete State-Space Problems with Explicit Validity Constraints","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/66R6IVCCSKXHQD3DCPQPRJOGZB","json":"https://pith.science/pith/66R6IVCCSKXHQD3DCPQPRJOGZB.json","graph_json":"https://pith.science/api/pith-number/66R6IVCCSKXHQD3DCPQPRJOGZB/graph.json","events_json":"https://pith.science/api/pith-number/66R6IVCCSKXHQD3DCPQPRJOGZB/events.json","paper":"https://pith.science/paper/66R6IVCC"},"agent_actions":{"view_html":"https://pith.science/pith/66R6IVCCSKXHQD3DCPQPRJOGZB","download_json":"https://pith.science/pith/66R6IVCCSKXHQD3DCPQPRJOGZB.json","view_paper":"https://pith.science/paper/66R6IVCC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.18099&json=true","fetch_graph":"https://pith.science/api/pith-number/66R6IVCCSKXHQD3DCPQPRJOGZB/graph.json","fetch_events":"https://pith.science/api/pith-number/66R6IVCCSKXHQD3DCPQPRJOGZB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/66R6IVCCSKXHQD3DCPQPRJOGZB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/66R6IVCCSKXHQD3DCPQPRJOGZB/action/storage_attestation","attest_author":"https://pith.science/pith/66R6IVCCSKXHQD3DCPQPRJOGZB/action/author_attestation","sign_citation":"https://pith.science/pith/66R6IVCCSKXHQD3DCPQPRJOGZB/action/citation_signature","submit_replication":"https://pith.science/pith/66R6IVCCSKXHQD3DCPQPRJOGZB/action/replication_record"}},"created_at":"2026-07-05T11:33:19.574787+00:00","updated_at":"2026-07-05T11:33:19.574787+00:00"}