{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T7OZNGU43EOKYEOYUHIH5B4LIP","short_pith_number":"pith:T7OZNGU4","schema_version":"1.0","canonical_sha256":"9fdd969a9cd91cac11d8a1d07e878b43f16c84582e9d40c8bfa376a8c401c142","source":{"kind":"arxiv","id":"2408.13338","version":1},"attestation_state":"computed","paper":{"title":"LalaEval: A Holistic Human Evaluation Framework for Domain-Specific Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.HC","authors_text":"Chengfei Fu, Chongyan Sun, Hulong Wu, Ken Lin, Shiwei Wang, Zhen Wang","submitted_at":"2024-08-23T19:12:45Z","abstract_excerpt":"This paper introduces LalaEval, a holistic framework designed for the human evaluation of domain-specific large language models (LLMs). LalaEval proposes a comprehensive suite of end-to-end protocols that cover five main components including domain specification, criteria establishment, benchmark dataset creation, construction of evaluation rubrics, and thorough analysis and interpretation of evaluation outcomes. This initiative aims to fill a crucial research gap by providing a systematic methodology for conducting standardized human evaluations within specific domains, a practice that, despi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.13338","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.HC","submitted_at":"2024-08-23T19:12:45Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"4dab71e0adf7e8d94c8b290a0ae4359d5f1a7b8c5049adb51e4789c662686480","abstract_canon_sha256":"5ab96f04a801588184f9adeb05ed4d34a5303eecc34d5df8b5d4edebf076fb4c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:58:49.037175Z","signature_b64":"3n1d/2ctkckluWgWYXjDnKWYMsBMLvxB4L3YYMiwgbLO/Yu7yHbVQY6GP+uPHXKLiq+uLPsVKmh82jUKn2V5DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9fdd969a9cd91cac11d8a1d07e878b43f16c84582e9d40c8bfa376a8c401c142","last_reissued_at":"2026-07-05T08:58:49.036695Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:58:49.036695Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LalaEval: A Holistic Human Evaluation Framework for Domain-Specific Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.HC","authors_text":"Chengfei Fu, Chongyan Sun, Hulong Wu, Ken Lin, Shiwei Wang, Zhen Wang","submitted_at":"2024-08-23T19:12:45Z","abstract_excerpt":"This paper introduces LalaEval, a holistic framework designed for the human evaluation of domain-specific large language models (LLMs). LalaEval proposes a comprehensive suite of end-to-end protocols that cover five main components including domain specification, criteria establishment, benchmark dataset creation, construction of evaluation rubrics, and thorough analysis and interpretation of evaluation outcomes. This initiative aims to fill a crucial research gap by providing a systematic methodology for conducting standardized human evaluations within specific domains, a practice that, despi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.13338","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.13338/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.13338","created_at":"2026-07-05T08:58:49.036761+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.13338v1","created_at":"2026-07-05T08:58:49.036761+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.13338","created_at":"2026-07-05T08:58:49.036761+00:00"},{"alias_kind":"pith_short_12","alias_value":"T7OZNGU43EOK","created_at":"2026-07-05T08:58:49.036761+00:00"},{"alias_kind":"pith_short_16","alias_value":"T7OZNGU43EOKYEOY","created_at":"2026-07-05T08:58:49.036761+00:00"},{"alias_kind":"pith_short_8","alias_value":"T7OZNGU4","created_at":"2026-07-05T08:58:49.036761+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.09670","citing_title":"The Science of Evaluating Foundation Models","ref_index":81,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T7OZNGU43EOKYEOYUHIH5B4LIP","json":"https://pith.science/pith/T7OZNGU43EOKYEOYUHIH5B4LIP.json","graph_json":"https://pith.science/api/pith-number/T7OZNGU43EOKYEOYUHIH5B4LIP/graph.json","events_json":"https://pith.science/api/pith-number/T7OZNGU43EOKYEOYUHIH5B4LIP/events.json","paper":"https://pith.science/paper/T7OZNGU4"},"agent_actions":{"view_html":"https://pith.science/pith/T7OZNGU43EOKYEOYUHIH5B4LIP","download_json":"https://pith.science/pith/T7OZNGU43EOKYEOYUHIH5B4LIP.json","view_paper":"https://pith.science/paper/T7OZNGU4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.13338&json=true","fetch_graph":"https://pith.science/api/pith-number/T7OZNGU43EOKYEOYUHIH5B4LIP/graph.json","fetch_events":"https://pith.science/api/pith-number/T7OZNGU43EOKYEOYUHIH5B4LIP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T7OZNGU43EOKYEOYUHIH5B4LIP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T7OZNGU43EOKYEOYUHIH5B4LIP/action/storage_attestation","attest_author":"https://pith.science/pith/T7OZNGU43EOKYEOYUHIH5B4LIP/action/author_attestation","sign_citation":"https://pith.science/pith/T7OZNGU43EOKYEOYUHIH5B4LIP/action/citation_signature","submit_replication":"https://pith.science/pith/T7OZNGU43EOKYEOYUHIH5B4LIP/action/replication_record"}},"created_at":"2026-07-05T08:58:49.036761+00:00","updated_at":"2026-07-05T08:58:49.036761+00:00"}