{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:WJLGMHSST2CNYPMMFJR7GL4BL5","short_pith_number":"pith:WJLGMHSS","canonical_record":{"source":{"id":"2401.16788","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-30T07:03:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9245dcb609f03a7b56c0bb6f2a6ab5216ed9323822dee923bac5d3817410b2e5","abstract_canon_sha256":"94093d9e1653298d63cd6c50a9a0d03f97b3167754180f40314598bf7d61d346"},"schema_version":"1.0"},"canonical_sha256":"b256661e529e84dc3d8c2a63f32f815f4bcc0b2373b4abf19ca5319f04798b58","source":{"kind":"arxiv","id":"2401.16788","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.16788","created_at":"2026-07-05T07:39:00Z"},{"alias_kind":"arxiv_version","alias_value":"2401.16788v1","created_at":"2026-07-05T07:39:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.16788","created_at":"2026-07-05T07:39:00Z"},{"alias_kind":"pith_short_12","alias_value":"WJLGMHSST2CN","created_at":"2026-07-05T07:39:00Z"},{"alias_kind":"pith_short_16","alias_value":"WJLGMHSST2CNYPMM","created_at":"2026-07-05T07:39:00Z"},{"alias_kind":"pith_short_8","alias_value":"WJLGMHSS","created_at":"2026-07-05T07:39:00Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:WJLGMHSST2CNYPMMFJR7GL4BL5","target":"record","payload":{"canonical_record":{"source":{"id":"2401.16788","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-30T07:03:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9245dcb609f03a7b56c0bb6f2a6ab5216ed9323822dee923bac5d3817410b2e5","abstract_canon_sha256":"94093d9e1653298d63cd6c50a9a0d03f97b3167754180f40314598bf7d61d346"},"schema_version":"1.0"},"canonical_sha256":"b256661e529e84dc3d8c2a63f32f815f4bcc0b2373b4abf19ca5319f04798b58","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:39:00.122500Z","signature_b64":"R69eAQpIeWMu7wrNDqOx5WlV7xTfuzcui1aUW7t8OfKjKhUSlTJuRlfOnV6XIpmDoW/HxFTXV3nKCx41Bw+sBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b256661e529e84dc3d8c2a63f32f815f4bcc0b2373b4abf19ca5319f04798b58","last_reissued_at":"2026-07-05T07:39:00.121998Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:39:00.121998Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2401.16788","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:39:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"gdHQmIxVS+HBXckoLlS6gTCIh5ufh5wJ124KddyIcwQae6TUzgN15OeDTtqHjOSTYriMbJH7qs59xyCd5l8wBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T14:50:42.717613Z"},"content_sha256":"1a81b2d1a413cb0b8be6fa75501d9e2916c856721704dd3193ea96dcf19a7036","schema_version":"1.0","event_id":"sha256:1a81b2d1a413cb0b8be6fa75501d9e2916c856721704dd3193ea96dcf19a7036"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:WJLGMHSST2CNYPMMFJR7GL4BL5","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Can Large Language Models be Trusted for Evaluation? Scalable Meta-Evaluation of LLMs as Evaluators via Agent Debate","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ethan Chern, Graham Neubig, Pengfei Liu, Steffi Chern","submitted_at":"2024-01-30T07:03:32Z","abstract_excerpt":"Despite the utility of Large Language Models (LLMs) across a wide range of tasks and scenarios, developing a method for reliably evaluating LLMs across varied contexts continues to be challenging. Modern evaluation approaches often use LLMs to assess responses generated by LLMs. However, the meta-evaluation conducted to assess the effectiveness of these LLMs as evaluators is typically constrained by the coverage of existing benchmarks or requires extensive human annotation. This underscores the urgency of methods for scalable meta-evaluation that can effectively, reliably, and efficiently eval"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.16788","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.16788/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:39:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Kx+Z2UaH2pBOiVdZHUXNK8qd+wSgyfhZGBqeCZuCh6I3DMKHCrTD9fcypXOplioSRtzx+1bcIRHagjDf4QjrBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T14:50:42.718534Z"},"content_sha256":"3e92a396139f723877c4836a190332aab305287ffc1eca3d4463c2cc31bba8c6","schema_version":"1.0","event_id":"sha256:3e92a396139f723877c4836a190332aab305287ffc1eca3d4463c2cc31bba8c6"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/WJLGMHSST2CNYPMMFJR7GL4BL5/bundle.json","state_url":"https://pith.science/pith/WJLGMHSST2CNYPMMFJR7GL4BL5/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/WJLGMHSST2CNYPMMFJR7GL4BL5/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T14:50:42Z","links":{"resolver":"https://pith.science/pith/WJLGMHSST2CNYPMMFJR7GL4BL5","bundle":"https://pith.science/pith/WJLGMHSST2CNYPMMFJR7GL4BL5/bundle.json","state":"https://pith.science/pith/WJLGMHSST2CNYPMMFJR7GL4BL5/state.json","well_known_bundle":"https://pith.science/.well-known/pith/WJLGMHSST2CNYPMMFJR7GL4BL5/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:WJLGMHSST2CNYPMMFJR7GL4BL5","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"94093d9e1653298d63cd6c50a9a0d03f97b3167754180f40314598bf7d61d346","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-30T07:03:32Z","title_canon_sha256":"9245dcb609f03a7b56c0bb6f2a6ab5216ed9323822dee923bac5d3817410b2e5"},"schema_version":"1.0","source":{"id":"2401.16788","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.16788","created_at":"2026-07-05T07:39:00Z"},{"alias_kind":"arxiv_version","alias_value":"2401.16788v1","created_at":"2026-07-05T07:39:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.16788","created_at":"2026-07-05T07:39:00Z"},{"alias_kind":"pith_short_12","alias_value":"WJLGMHSST2CN","created_at":"2026-07-05T07:39:00Z"},{"alias_kind":"pith_short_16","alias_value":"WJLGMHSST2CNYPMM","created_at":"2026-07-05T07:39:00Z"},{"alias_kind":"pith_short_8","alias_value":"WJLGMHSS","created_at":"2026-07-05T07:39:00Z"}],"graph_snapshots":[{"event_id":"sha256:3e92a396139f723877c4836a190332aab305287ffc1eca3d4463c2cc31bba8c6","target":"graph","created_at":"2026-07-05T07:39:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2401.16788/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Despite the utility of Large Language Models (LLMs) across a wide range of tasks and scenarios, developing a method for reliably evaluating LLMs across varied contexts continues to be challenging. Modern evaluation approaches often use LLMs to assess responses generated by LLMs. However, the meta-evaluation conducted to assess the effectiveness of these LLMs as evaluators is typically constrained by the coverage of existing benchmarks or requires extensive human annotation. This underscores the urgency of methods for scalable meta-evaluation that can effectively, reliably, and efficiently eval","authors_text":"Ethan Chern, Graham Neubig, Pengfei Liu, Steffi Chern","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-30T07:03:32Z","title":"Can Large Language Models be Trusted for Evaluation? Scalable Meta-Evaluation of LLMs as Evaluators via Agent Debate"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.16788","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1a81b2d1a413cb0b8be6fa75501d9e2916c856721704dd3193ea96dcf19a7036","target":"record","created_at":"2026-07-05T07:39:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"94093d9e1653298d63cd6c50a9a0d03f97b3167754180f40314598bf7d61d346","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-30T07:03:32Z","title_canon_sha256":"9245dcb609f03a7b56c0bb6f2a6ab5216ed9323822dee923bac5d3817410b2e5"},"schema_version":"1.0","source":{"id":"2401.16788","kind":"arxiv","version":1}},"canonical_sha256":"b256661e529e84dc3d8c2a63f32f815f4bcc0b2373b4abf19ca5319f04798b58","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b256661e529e84dc3d8c2a63f32f815f4bcc0b2373b4abf19ca5319f04798b58","first_computed_at":"2026-07-05T07:39:00.121998Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:39:00.121998Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"R69eAQpIeWMu7wrNDqOx5WlV7xTfuzcui1aUW7t8OfKjKhUSlTJuRlfOnV6XIpmDoW/HxFTXV3nKCx41Bw+sBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T07:39:00.122500Z","signed_message":"canonical_sha256_bytes"},"source_id":"2401.16788","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1a81b2d1a413cb0b8be6fa75501d9e2916c856721704dd3193ea96dcf19a7036","sha256:3e92a396139f723877c4836a190332aab305287ffc1eca3d4463c2cc31bba8c6"],"state_sha256":"1ea78a3ce1b09b7b9d39fe87e42a587dd51280e628608fc3c85e77a8360d739e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"YBjAFzrHUCRUPCORVq4RADW0dVaCc9dGbXgWz1+DiSFqbddVdLl70cBMO9bbW+Y7fSQ26+IYTMNWuc0FJt1fDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T14:50:42.725586Z","bundle_sha256":"0bc49b8542cc16fd0d21232af236a93fe18bd9efebb160dc9cf502446d19a478"}}