{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GBGRXUQNQ3RMSGSWEGVUV7HLPD","short_pith_number":"pith:GBGRXUQN","schema_version":"1.0","canonical_sha256":"304d1bd20d86e2c91a5621ab4afceb78f7305b0bc168fd54a015fb361d4831d0","source":{"kind":"arxiv","id":"2405.01535","version":2},"attestation_state":"computed","paper":{"title":"Prometheus 2: An Open Source Language Model Specialized in Evaluating Other Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bill Yuchen Lin, Graham Neubig, Jamin Shin, Juyoung Suk, Kyungjae Lee, Minjoon Seo, Moontae Lee, Sean Welleck, Seungone Kim, Shayne Longpre","submitted_at":"2024-05-02T17:59:35Z","abstract_excerpt":"Proprietary LMs such as GPT-4 are often employed to assess the quality of responses from various LMs. However, concerns including transparency, controllability, and affordability strongly motivate the development of open-source LMs specialized in evaluations. On the other hand, existing open evaluator LMs exhibit critical shortcomings: 1) they issue scores that significantly diverge from those assigned by humans, and 2) they lack the flexibility to perform both direct assessment and pairwise ranking, the two most prevalent forms of assessment. Additionally, they do not possess the ability to e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.01535","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-02T17:59:35Z","cross_cats_sorted":[],"title_canon_sha256":"0e1c5998d6f70bfc614b0801c8de4ab32d013b641c65c7768656ca6124462267","abstract_canon_sha256":"7ca089558442566043a8874efa3ff54465ddb4068634b108c43fd812d5df5647"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:44:34.247241Z","signature_b64":"eofJJksHaYBYmyoLLRZSv/aCigul9Lri6MPrOkZ2/+2Setp7aqreTKNaz1OVrfVwdjOm4vCSMBpwUqA5Vs6GDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"304d1bd20d86e2c91a5621ab4afceb78f7305b0bc168fd54a015fb361d4831d0","last_reissued_at":"2026-07-05T09:44:34.246740Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:44:34.246740Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prometheus 2: An Open Source Language Model Specialized in Evaluating Other Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bill Yuchen Lin, Graham Neubig, Jamin Shin, Juyoung Suk, Kyungjae Lee, Minjoon Seo, Moontae Lee, Sean Welleck, Seungone Kim, Shayne Longpre","submitted_at":"2024-05-02T17:59:35Z","abstract_excerpt":"Proprietary LMs such as GPT-4 are often employed to assess the quality of responses from various LMs. However, concerns including transparency, controllability, and affordability strongly motivate the development of open-source LMs specialized in evaluations. On the other hand, existing open evaluator LMs exhibit critical shortcomings: 1) they issue scores that significantly diverge from those assigned by humans, and 2) they lack the flexibility to perform both direct assessment and pairwise ranking, the two most prevalent forms of assessment. Additionally, they do not possess the ability to e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.01535","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.01535/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.01535","created_at":"2026-07-05T09:44:34.246799+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.01535v2","created_at":"2026-07-05T09:44:34.246799+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.01535","created_at":"2026-07-05T09:44:34.246799+00:00"},{"alias_kind":"pith_short_12","alias_value":"GBGRXUQNQ3RM","created_at":"2026-07-05T09:44:34.246799+00:00"},{"alias_kind":"pith_short_16","alias_value":"GBGRXUQNQ3RMSGSW","created_at":"2026-07-05T09:44:34.246799+00:00"},{"alias_kind":"pith_short_8","alias_value":"GBGRXUQN","created_at":"2026-07-05T09:44:34.246799+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2607.06641","citing_title":"Healthier LLMs: Retrieval-Augmented Generation for Public Health Question Answering","ref_index":41,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08535","citing_title":"When the Judge Changes, So Does the Measurement: Auditing LLM-as-Judge Reliability","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11196","citing_title":"PoQ-Judge: A Multi-Architecture Evaluation Framework for Cost-Aware Proof-of-Quality in Decentralized LLM Inference","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23583","citing_title":"Evaluation Awareness Is Not One Capability: Evidence from Open Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19057","citing_title":"Quantifying and Auditing LLM Evaluation via Positive--Unlabeled Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07040","citing_title":"Beyond Rubrics: Exploration-Guided Evaluation Skills for Reward Modeling","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04306","citing_title":"Organizational Control Layer: Governance Infrastructure at the Execution Boundary of LLM Agent Systems","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03650","citing_title":"CoEval: Ranking Language Models for Custom Tasks Without Labeled Data or Trustworthy Benchmarks","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30931","citing_title":"RoPoLL: Robust Panel of LLM Judges","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20608","citing_title":"CourseBlueprint: A Structured Pipeline for Adaptive Pedagogical Video Generation Grounded in Course Corpora","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25038","citing_title":"TRACE: A taxonomy-grounded synthetic dataset for teaching-program generation and session interpretation in Applied Behavior Analysis","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30116","citing_title":"Open Problems in Constitutional Preference Reconstruction","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29522","citing_title":"DeepSurvey: Enhancing Analytical Depth and Citation Reliability in Automated Survey Generation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02443","citing_title":"HalluScan: A Systematic Benchmark for Detecting and Mitigating Hallucinations in Instruction-Following LLMs","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21319","citing_title":"RLBFF: Binary Flexible Feedback to bridge between Human Feedback & Verifiable Rewards","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23542","citing_title":"On the Shelf Life of Fine-Tuned LLM-Judges: Future-Proofing, Backward-Compatibility, and Question Generalization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23213","citing_title":"Scoring, Reasoning, and Selecting the Best! Ensembling Large Language Models via a Peer-Review Process","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16335","citing_title":"Beyond Verifiable Rewards: Rubric-Based GRM for Reinforced Fine-Tuning SWE Agents","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2507.17746","citing_title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23178","citing_title":"Judging the Judges: A Systematic Evaluation of Bias Mitigation Strategies in LLM-as-a-Judge Pipelines","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04785","citing_title":"AgentTrust: Runtime Safety Evaluation and Interception for AI Agent Tool Use","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08061","citing_title":"Rubric-Grounded RL: Structured Judge Rewards for Generalizable Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14585","citing_title":"Prompt Optimization Is a Coin Flip: Diagnosing When It Helps in Compound AI Systems","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19820","citing_title":"KnowPilot: Your Knowledge-Driven Copilot for Domain Tasks","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GBGRXUQNQ3RMSGSWEGVUV7HLPD","json":"https://pith.science/pith/GBGRXUQNQ3RMSGSWEGVUV7HLPD.json","graph_json":"https://pith.science/api/pith-number/GBGRXUQNQ3RMSGSWEGVUV7HLPD/graph.json","events_json":"https://pith.science/api/pith-number/GBGRXUQNQ3RMSGSWEGVUV7HLPD/events.json","paper":"https://pith.science/paper/GBGRXUQN"},"agent_actions":{"view_html":"https://pith.science/pith/GBGRXUQNQ3RMSGSWEGVUV7HLPD","download_json":"https://pith.science/pith/GBGRXUQNQ3RMSGSWEGVUV7HLPD.json","view_paper":"https://pith.science/paper/GBGRXUQN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.01535&json=true","fetch_graph":"https://pith.science/api/pith-number/GBGRXUQNQ3RMSGSWEGVUV7HLPD/graph.json","fetch_events":"https://pith.science/api/pith-number/GBGRXUQNQ3RMSGSWEGVUV7HLPD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GBGRXUQNQ3RMSGSWEGVUV7HLPD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GBGRXUQNQ3RMSGSWEGVUV7HLPD/action/storage_attestation","attest_author":"https://pith.science/pith/GBGRXUQNQ3RMSGSWEGVUV7HLPD/action/author_attestation","sign_citation":"https://pith.science/pith/GBGRXUQNQ3RMSGSWEGVUV7HLPD/action/citation_signature","submit_replication":"https://pith.science/pith/GBGRXUQNQ3RMSGSWEGVUV7HLPD/action/replication_record"}},"created_at":"2026-07-05T09:44:34.246799+00:00","updated_at":"2026-07-05T09:44:34.246799+00:00"}