{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XCALAXG6M7E2TP4GJJNN6BYSWS","short_pith_number":"pith:XCALAXG6","schema_version":"1.0","canonical_sha256":"b880b05cde67c9a9bf864a5adf0712b4ac2beb3168bdb6dd0185b4bc9c367804","source":{"kind":"arxiv","id":"2407.10817","version":1},"attestation_state":"computed","paper":{"title":"Foundational Autoraters: Taming Large Language Models for Better Automatic Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chris Tar, Kalpesh Krishna, Manaal Faruqui, Salaheddin Alzubi, Tu Vu, Yun-hsuan Sung","submitted_at":"2024-07-15T15:33:45Z","abstract_excerpt":"As large language models (LLMs) advance, it becomes more challenging to reliably evaluate their output due to the high costs of human evaluation. To make progress towards better LLM autoraters, we introduce FLAMe, a family of Foundational Large Autorater Models. FLAMe is trained on our large and diverse collection of 100+ quality assessment tasks comprising 5M+ human judgments, curated and standardized using publicly released human evaluations from previous research. FLAMe significantly improves generalization to a wide variety of held-out tasks, outperforming LLMs trained on proprietary data "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.10817","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-15T15:33:45Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"054d86dd07eab7a0d7069b06c8aa44fd0cd7f304021a382831903052c1c4ec98","abstract_canon_sha256":"25736d1a5d2bdc952d39a840276f87f3b9d97f3cbb4a6c6df66bf6e2b1f3df69"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:04.606838Z","signature_b64":"TP5FmsmVwytLkTzjyLMm5b2r805tWb0Lr8ALig/v1mD8obf+LIcNfA1AVC0rVHi0gW5iu7oQdFvk1EFGmz7bCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b880b05cde67c9a9bf864a5adf0712b4ac2beb3168bdb6dd0185b4bc9c367804","last_reissued_at":"2026-07-05T08:44:04.606324Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:04.606324Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Foundational Autoraters: Taming Large Language Models for Better Automatic Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chris Tar, Kalpesh Krishna, Manaal Faruqui, Salaheddin Alzubi, Tu Vu, Yun-hsuan Sung","submitted_at":"2024-07-15T15:33:45Z","abstract_excerpt":"As large language models (LLMs) advance, it becomes more challenging to reliably evaluate their output due to the high costs of human evaluation. To make progress towards better LLM autoraters, we introduce FLAMe, a family of Foundational Large Autorater Models. FLAMe is trained on our large and diverse collection of 100+ quality assessment tasks comprising 5M+ human judgments, curated and standardized using publicly released human evaluations from previous research. FLAMe significantly improves generalization to a wide variety of held-out tasks, outperforming LLMs trained on proprietary data "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.10817","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.10817/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.10817","created_at":"2026-07-05T08:44:04.606387+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.10817v1","created_at":"2026-07-05T08:44:04.606387+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.10817","created_at":"2026-07-05T08:44:04.606387+00:00"},{"alias_kind":"pith_short_12","alias_value":"XCALAXG6M7E2","created_at":"2026-07-05T08:44:04.606387+00:00"},{"alias_kind":"pith_short_16","alias_value":"XCALAXG6M7E2TP4G","created_at":"2026-07-05T08:44:04.606387+00:00"},{"alias_kind":"pith_short_8","alias_value":"XCALAXG6","created_at":"2026-07-05T08:44:04.606387+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03650","citing_title":"CoEval: Ranking Language Models for Custom Tasks Without Labeled Data or Trustworthy Benchmarks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2508.19932","citing_title":"CASE: An Agentic AI Framework for Enhancing Scam Intelligence in Digital Payments","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2603.02766","citing_title":"EvoSkill: Automated Skill Discovery for Multi-Agent Systems","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":234,"is_internal_anchor":false},{"citing_arxiv_id":"2502.18864","citing_title":"Towards an AI co-scientist","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XCALAXG6M7E2TP4GJJNN6BYSWS","json":"https://pith.science/pith/XCALAXG6M7E2TP4GJJNN6BYSWS.json","graph_json":"https://pith.science/api/pith-number/XCALAXG6M7E2TP4GJJNN6BYSWS/graph.json","events_json":"https://pith.science/api/pith-number/XCALAXG6M7E2TP4GJJNN6BYSWS/events.json","paper":"https://pith.science/paper/XCALAXG6"},"agent_actions":{"view_html":"https://pith.science/pith/XCALAXG6M7E2TP4GJJNN6BYSWS","download_json":"https://pith.science/pith/XCALAXG6M7E2TP4GJJNN6BYSWS.json","view_paper":"https://pith.science/paper/XCALAXG6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.10817&json=true","fetch_graph":"https://pith.science/api/pith-number/XCALAXG6M7E2TP4GJJNN6BYSWS/graph.json","fetch_events":"https://pith.science/api/pith-number/XCALAXG6M7E2TP4GJJNN6BYSWS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XCALAXG6M7E2TP4GJJNN6BYSWS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XCALAXG6M7E2TP4GJJNN6BYSWS/action/storage_attestation","attest_author":"https://pith.science/pith/XCALAXG6M7E2TP4GJJNN6BYSWS/action/author_attestation","sign_citation":"https://pith.science/pith/XCALAXG6M7E2TP4GJJNN6BYSWS/action/citation_signature","submit_replication":"https://pith.science/pith/XCALAXG6M7E2TP4GJJNN6BYSWS/action/replication_record"}},"created_at":"2026-07-05T08:44:04.606387+00:00","updated_at":"2026-07-05T08:44:04.606387+00:00"}