{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:W65SF6X2Z55UAJ2TXLW5QPZYVM","short_pith_number":"pith:W65SF6X2","schema_version":"1.0","canonical_sha256":"b7bb22fafacf7b402753baedd83f38ab18426fd5e90ffdb165250358b44a5181","source":{"kind":"arxiv","id":"2410.14044","version":1},"attestation_state":"computed","paper":{"title":"Best in Tau@LLMJudge: Criteria-Based Relevance Evaluation with Llama3","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.IR","authors_text":"Laura Dietz, Naghmeh Farzi","submitted_at":"2024-10-17T21:37:08Z","abstract_excerpt":"Traditional evaluation of information retrieval (IR) systems relies on human-annotated relevance labels, which can be both biased and costly at scale. In this context, large language models (LLMs) offer an alternative by allowing us to directly prompt them to assign relevance labels for passages associated with each query. In this study, we explore alternative methods to directly prompt LLMs for assigned relevance labels, by exploring two hypotheses:\n  Hypothesis 1 assumes that it is helpful to break down \"relevance\" into specific criteria - exactness, coverage, topicality, and contextual fit."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.14044","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2024-10-17T21:37:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"68d1c577fb96e7df0dcfb27e84bf15d929ec6240731a249ff67f0c6f86557492","abstract_canon_sha256":"f2fd849f44041c4c90dc9a3b7875a83f32c415d9289dfca1dad56b05e4fc7621"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:22:26.331801Z","signature_b64":"QQ/6/1yJxgEVLTUbHIwaIdJMagQpVIcdatuFJr5qUJKhrqgVqs5Vw6EYsuSMrLTLOnsQFlyB7ZSg2uScOEqVAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b7bb22fafacf7b402753baedd83f38ab18426fd5e90ffdb165250358b44a5181","last_reissued_at":"2026-07-05T09:22:26.331391Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:22:26.331391Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Best in Tau@LLMJudge: Criteria-Based Relevance Evaluation with Llama3","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.IR","authors_text":"Laura Dietz, Naghmeh Farzi","submitted_at":"2024-10-17T21:37:08Z","abstract_excerpt":"Traditional evaluation of information retrieval (IR) systems relies on human-annotated relevance labels, which can be both biased and costly at scale. In this context, large language models (LLMs) offer an alternative by allowing us to directly prompt them to assign relevance labels for passages associated with each query. In this study, we explore alternative methods to directly prompt LLMs for assigned relevance labels, by exploring two hypotheses:\n  Hypothesis 1 assumes that it is helpful to break down \"relevance\" into specific criteria - exactness, coverage, topicality, and contextual fit."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.14044","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.14044/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.14044","created_at":"2026-07-05T09:22:26.331452+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.14044v1","created_at":"2026-07-05T09:22:26.331452+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.14044","created_at":"2026-07-05T09:22:26.331452+00:00"},{"alias_kind":"pith_short_12","alias_value":"W65SF6X2Z55U","created_at":"2026-07-05T09:22:26.331452+00:00"},{"alias_kind":"pith_short_16","alias_value":"W65SF6X2Z55UAJ2T","created_at":"2026-07-05T09:22:26.331452+00:00"},{"alias_kind":"pith_short_8","alias_value":"W65SF6X2","created_at":"2026-07-05T09:22:26.331452+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.13268","citing_title":"JudgeBlender: Ensembling Judgments for Automatic Relevance Assessment","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W65SF6X2Z55UAJ2TXLW5QPZYVM","json":"https://pith.science/pith/W65SF6X2Z55UAJ2TXLW5QPZYVM.json","graph_json":"https://pith.science/api/pith-number/W65SF6X2Z55UAJ2TXLW5QPZYVM/graph.json","events_json":"https://pith.science/api/pith-number/W65SF6X2Z55UAJ2TXLW5QPZYVM/events.json","paper":"https://pith.science/paper/W65SF6X2"},"agent_actions":{"view_html":"https://pith.science/pith/W65SF6X2Z55UAJ2TXLW5QPZYVM","download_json":"https://pith.science/pith/W65SF6X2Z55UAJ2TXLW5QPZYVM.json","view_paper":"https://pith.science/paper/W65SF6X2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.14044&json=true","fetch_graph":"https://pith.science/api/pith-number/W65SF6X2Z55UAJ2TXLW5QPZYVM/graph.json","fetch_events":"https://pith.science/api/pith-number/W65SF6X2Z55UAJ2TXLW5QPZYVM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W65SF6X2Z55UAJ2TXLW5QPZYVM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W65SF6X2Z55UAJ2TXLW5QPZYVM/action/storage_attestation","attest_author":"https://pith.science/pith/W65SF6X2Z55UAJ2TXLW5QPZYVM/action/author_attestation","sign_citation":"https://pith.science/pith/W65SF6X2Z55UAJ2TXLW5QPZYVM/action/citation_signature","submit_replication":"https://pith.science/pith/W65SF6X2Z55UAJ2TXLW5QPZYVM/action/replication_record"}},"created_at":"2026-07-05T09:22:26.331452+00:00","updated_at":"2026-07-05T09:22:26.331452+00:00"}