{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:2BLUN7DKX5W2PG3D7KDQ2HUSYF","short_pith_number":"pith:2BLUN7DK","canonical_record":{"source":{"id":"2404.01667","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-02T06:14:54Z","cross_cats_sorted":[],"title_canon_sha256":"91cf4990545fd3ad6490a4fed2c49abf56101666951f1d4c45c6c370062ebbba","abstract_canon_sha256":"8ff1a15ce4f9fd3e34dea443fe3438a63ada7e901fb67246b670e30c593a9dcd"},"schema_version":"1.0"},"canonical_sha256":"d05746fc6abf6da79b63fa870d1e92c15982ab13551141f13989c84f8cc1ca23","source":{"kind":"arxiv","id":"2404.01667","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.01667","created_at":"2026-07-05T08:03:17Z"},{"alias_kind":"arxiv_version","alias_value":"2404.01667v1","created_at":"2026-07-05T08:03:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.01667","created_at":"2026-07-05T08:03:17Z"},{"alias_kind":"pith_short_12","alias_value":"2BLUN7DKX5W2","created_at":"2026-07-05T08:03:17Z"},{"alias_kind":"pith_short_16","alias_value":"2BLUN7DKX5W2PG3D","created_at":"2026-07-05T08:03:17Z"},{"alias_kind":"pith_short_8","alias_value":"2BLUN7DK","created_at":"2026-07-05T08:03:17Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:2BLUN7DKX5W2PG3D7KDQ2HUSYF","target":"record","payload":{"canonical_record":{"source":{"id":"2404.01667","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-02T06:14:54Z","cross_cats_sorted":[],"title_canon_sha256":"91cf4990545fd3ad6490a4fed2c49abf56101666951f1d4c45c6c370062ebbba","abstract_canon_sha256":"8ff1a15ce4f9fd3e34dea443fe3438a63ada7e901fb67246b670e30c593a9dcd"},"schema_version":"1.0"},"canonical_sha256":"d05746fc6abf6da79b63fa870d1e92c15982ab13551141f13989c84f8cc1ca23","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:03:17.951568Z","signature_b64":"g8LquMbdegbimODrPlGeFC48EntyU3MPq6RktEbhw4C2lmitfYIwPZqC1o2PoB6AqJW8sG5xYcnkdLcXvbilCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d05746fc6abf6da79b63fa870d1e92c15982ab13551141f13989c84f8cc1ca23","last_reissued_at":"2026-07-05T08:03:17.951139Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:03:17.951139Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2404.01667","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:03:17Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"06Hf81JfzgW7/elK66Ef5ayPfnANs6b7Jm8RsAhMuRKgF/eHpwzWtbbIoTHcMsMreSDR4OE28KpE0NvUHk9sAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T10:40:49.629745Z"},"content_sha256":"4c7d42eff6fc5a847554104779371cf0aea8a29bcf98c22f7f1719f175110e37","schema_version":"1.0","event_id":"sha256:4c7d42eff6fc5a847554104779371cf0aea8a29bcf98c22f7f1719f175110e37"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:2BLUN7DKX5W2PG3D7KDQ2HUSYF","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"METAL: Towards Multilingual Meta-Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kalika Bali, Mohamed Ahmed, Rishav Hada, Sunayana Sitaram, Varun Gumma","submitted_at":"2024-04-02T06:14:54Z","abstract_excerpt":"With the rising human-like precision of Large Language Models (LLMs) in numerous tasks, their utilization in a variety of real-world applications is becoming more prevalent. Several studies have shown that LLMs excel on many standard NLP benchmarks. However, it is challenging to evaluate LLMs due to test dataset contamination and the limitations of traditional metrics. Since human evaluations are difficult to collect, there is a growing interest in the community to use LLMs themselves as reference-free evaluators for subjective metrics. However, past work has shown that LLM-based evaluators ca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.01667","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.01667/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:03:17Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"B9juSb0/zvd3bM/Q/HuPu3wWXu58f3MFyp4PO3VforBSyVXTuUq/RHHGIR5orK2koi0A5PuVB2WKYg7ts0uaCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T10:40:49.630261Z"},"content_sha256":"561628e46d83b8405fe67bd6efa5601fbca299d588ee1861313fddae026ba0ad","schema_version":"1.0","event_id":"sha256:561628e46d83b8405fe67bd6efa5601fbca299d588ee1861313fddae026ba0ad"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2BLUN7DKX5W2PG3D7KDQ2HUSYF/bundle.json","state_url":"https://pith.science/pith/2BLUN7DKX5W2PG3D7KDQ2HUSYF/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2BLUN7DKX5W2PG3D7KDQ2HUSYF/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-19T10:40:49Z","links":{"resolver":"https://pith.science/pith/2BLUN7DKX5W2PG3D7KDQ2HUSYF","bundle":"https://pith.science/pith/2BLUN7DKX5W2PG3D7KDQ2HUSYF/bundle.json","state":"https://pith.science/pith/2BLUN7DKX5W2PG3D7KDQ2HUSYF/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2BLUN7DKX5W2PG3D7KDQ2HUSYF/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:2BLUN7DKX5W2PG3D7KDQ2HUSYF","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8ff1a15ce4f9fd3e34dea443fe3438a63ada7e901fb67246b670e30c593a9dcd","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-02T06:14:54Z","title_canon_sha256":"91cf4990545fd3ad6490a4fed2c49abf56101666951f1d4c45c6c370062ebbba"},"schema_version":"1.0","source":{"id":"2404.01667","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.01667","created_at":"2026-07-05T08:03:17Z"},{"alias_kind":"arxiv_version","alias_value":"2404.01667v1","created_at":"2026-07-05T08:03:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.01667","created_at":"2026-07-05T08:03:17Z"},{"alias_kind":"pith_short_12","alias_value":"2BLUN7DKX5W2","created_at":"2026-07-05T08:03:17Z"},{"alias_kind":"pith_short_16","alias_value":"2BLUN7DKX5W2PG3D","created_at":"2026-07-05T08:03:17Z"},{"alias_kind":"pith_short_8","alias_value":"2BLUN7DK","created_at":"2026-07-05T08:03:17Z"}],"graph_snapshots":[{"event_id":"sha256:561628e46d83b8405fe67bd6efa5601fbca299d588ee1861313fddae026ba0ad","target":"graph","created_at":"2026-07-05T08:03:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2404.01667/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"With the rising human-like precision of Large Language Models (LLMs) in numerous tasks, their utilization in a variety of real-world applications is becoming more prevalent. Several studies have shown that LLMs excel on many standard NLP benchmarks. However, it is challenging to evaluate LLMs due to test dataset contamination and the limitations of traditional metrics. Since human evaluations are difficult to collect, there is a growing interest in the community to use LLMs themselves as reference-free evaluators for subjective metrics. However, past work has shown that LLM-based evaluators ca","authors_text":"Kalika Bali, Mohamed Ahmed, Rishav Hada, Sunayana Sitaram, Varun Gumma","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-02T06:14:54Z","title":"METAL: Towards Multilingual Meta-Evaluation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.01667","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4c7d42eff6fc5a847554104779371cf0aea8a29bcf98c22f7f1719f175110e37","target":"record","created_at":"2026-07-05T08:03:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8ff1a15ce4f9fd3e34dea443fe3438a63ada7e901fb67246b670e30c593a9dcd","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-02T06:14:54Z","title_canon_sha256":"91cf4990545fd3ad6490a4fed2c49abf56101666951f1d4c45c6c370062ebbba"},"schema_version":"1.0","source":{"id":"2404.01667","kind":"arxiv","version":1}},"canonical_sha256":"d05746fc6abf6da79b63fa870d1e92c15982ab13551141f13989c84f8cc1ca23","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d05746fc6abf6da79b63fa870d1e92c15982ab13551141f13989c84f8cc1ca23","first_computed_at":"2026-07-05T08:03:17.951139Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:03:17.951139Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"g8LquMbdegbimODrPlGeFC48EntyU3MPq6RktEbhw4C2lmitfYIwPZqC1o2PoB6AqJW8sG5xYcnkdLcXvbilCw==","signature_status":"signed_v1","signed_at":"2026-07-05T08:03:17.951568Z","signed_message":"canonical_sha256_bytes"},"source_id":"2404.01667","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4c7d42eff6fc5a847554104779371cf0aea8a29bcf98c22f7f1719f175110e37","sha256:561628e46d83b8405fe67bd6efa5601fbca299d588ee1861313fddae026ba0ad"],"state_sha256":"c9a3bf488fa38a2ab2eac15c6b0938842f821cb1934f0d7cbfa7ee0fd3f3e0e5"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"lCAGRg/utpm1Gcty+dNUgu6JOF+Bw+P28CI5YkDq2cAVfvGmK0tM4XqhPgViKcc49vvisdtrhMhNXYDBKVjkCw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-19T10:40:49.635245Z","bundle_sha256":"cd8995da64ad8502281397b5457b8863cb47326a62f18ca83415bee2bdb52e0f"}}