{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:72UO6WMR6BUVKJW25YCGWDKMKA","short_pith_number":"pith:72UO6WMR","schema_version":"1.0","canonical_sha256":"fea8ef5991f0695526daee046b0d4c5000f567a34c01829894c737c179ed565a","source":{"kind":"arxiv","id":"2502.04997","version":1},"attestation_state":"computed","paper":{"title":"Aligning Black-box Language Models with Human Judgments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Gen Suzuki, Gerrit J. J. van den Burg, Murat Sensoy, Wei Liu","submitted_at":"2025-02-07T15:19:40Z","abstract_excerpt":"Large language models (LLMs) are increasingly used as automated judges to evaluate recommendation systems, search engines, and other subjective tasks, where relying on human evaluators can be costly, time-consuming, and unscalable. LLMs offer an efficient solution for continuous, automated evaluation. However, since the systems that are built and improved with these judgments are ultimately designed for human use, it is crucial that LLM judgments align closely with human evaluators to ensure such systems remain human-centered. On the other hand, aligning LLM judgments with human evaluators is "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.04997","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-07T15:19:40Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"391eb350cb7b8150b8e9636da52d4d0a6affcce22bb411175ff8873253e8c8d0","abstract_canon_sha256":"385899a3f02edd6da017b04232b790c852cf6ce2e539e97907fb435f1fe95990"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:01.144682Z","signature_b64":"VyZ4CPmNaADKWWxSqL6ayWjvFrva8rU57OgdrtltvjDo566sSo3J+XR2MziNQulJF7rhA1rHKgQTmEov5x6OCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fea8ef5991f0695526daee046b0d4c5000f567a34c01829894c737c179ed565a","last_reissued_at":"2026-07-05T10:11:01.144154Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:01.144154Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Aligning Black-box Language Models with Human Judgments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Gen Suzuki, Gerrit J. J. van den Burg, Murat Sensoy, Wei Liu","submitted_at":"2025-02-07T15:19:40Z","abstract_excerpt":"Large language models (LLMs) are increasingly used as automated judges to evaluate recommendation systems, search engines, and other subjective tasks, where relying on human evaluators can be costly, time-consuming, and unscalable. LLMs offer an efficient solution for continuous, automated evaluation. However, since the systems that are built and improved with these judgments are ultimately designed for human use, it is crucial that LLM judgments align closely with human evaluators to ensure such systems remain human-centered. On the other hand, aligning LLM judgments with human evaluators is "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.04997","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.04997/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.04997","created_at":"2026-07-05T10:11:01.144225+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.04997v1","created_at":"2026-07-05T10:11:01.144225+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.04997","created_at":"2026-07-05T10:11:01.144225+00:00"},{"alias_kind":"pith_short_12","alias_value":"72UO6WMR6BUV","created_at":"2026-07-05T10:11:01.144225+00:00"},{"alias_kind":"pith_short_16","alias_value":"72UO6WMR6BUVKJW2","created_at":"2026-07-05T10:11:01.144225+00:00"},{"alias_kind":"pith_short_8","alias_value":"72UO6WMR","created_at":"2026-07-05T10:11:01.144225+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/72UO6WMR6BUVKJW25YCGWDKMKA","json":"https://pith.science/pith/72UO6WMR6BUVKJW25YCGWDKMKA.json","graph_json":"https://pith.science/api/pith-number/72UO6WMR6BUVKJW25YCGWDKMKA/graph.json","events_json":"https://pith.science/api/pith-number/72UO6WMR6BUVKJW25YCGWDKMKA/events.json","paper":"https://pith.science/paper/72UO6WMR"},"agent_actions":{"view_html":"https://pith.science/pith/72UO6WMR6BUVKJW25YCGWDKMKA","download_json":"https://pith.science/pith/72UO6WMR6BUVKJW25YCGWDKMKA.json","view_paper":"https://pith.science/paper/72UO6WMR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.04997&json=true","fetch_graph":"https://pith.science/api/pith-number/72UO6WMR6BUVKJW25YCGWDKMKA/graph.json","fetch_events":"https://pith.science/api/pith-number/72UO6WMR6BUVKJW25YCGWDKMKA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/72UO6WMR6BUVKJW25YCGWDKMKA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/72UO6WMR6BUVKJW25YCGWDKMKA/action/storage_attestation","attest_author":"https://pith.science/pith/72UO6WMR6BUVKJW25YCGWDKMKA/action/author_attestation","sign_citation":"https://pith.science/pith/72UO6WMR6BUVKJW25YCGWDKMKA/action/citation_signature","submit_replication":"https://pith.science/pith/72UO6WMR6BUVKJW25YCGWDKMKA/action/replication_record"}},"created_at":"2026-07-05T10:11:01.144225+00:00","updated_at":"2026-07-05T10:11:01.144225+00:00"}