{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FPL5SNEUQEJT7EWRWGW6UIRPDW","short_pith_number":"pith:FPL5SNEU","schema_version":"1.0","canonical_sha256":"2bd7d9349481133f92d1b1adea222f1db64101a481528a69b31f3398e650c09e","source":{"kind":"arxiv","id":"2406.16203","version":3},"attestation_state":"computed","paper":{"title":"LLMs' Classification Performance is Overclaimed","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Elizabeth Garrison, Elmira Talebianaraki, Hanzi Xu, Jiangshu Du, Renze Lou, Slobodan Vucetic, Vahid Mahzoon, Wenpeng Yin, Zhuoan Zhou","submitted_at":"2024-06-23T19:49:10Z","abstract_excerpt":"In many classification tasks designed for AI or human to solve, gold labels are typically included within the label space by default, often posed as \"which of the following is correct?\" This standard setup has traditionally highlighted the strong performance of advanced AI, particularly top-performing Large Language Models (LLMs), in routine classification tasks. However, when the gold label is intentionally excluded from the label space, it becomes evident that LLMs still attempt to select from the available label candidates, even when none are correct. This raises a pivotal question: Do LLMs"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.16203","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-23T19:49:10Z","cross_cats_sorted":[],"title_canon_sha256":"be44bee7298288d86c3915a538f16efbbe893260ba54f165ddc8aed437cd6abc","abstract_canon_sha256":"e2209f3742ec328d19e1aebbdafccb44266e7d718101b7caa5a05f48857935b4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:39:35.392562Z","signature_b64":"LlvKwyjBhPHdtfBEC+ayGK6ovKEpA1c3Ky7+koSWT1yp77Jis5LzPu0Z399hlnnTqeqA0jBPBGUNNq0Y18a7CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2bd7d9349481133f92d1b1adea222f1db64101a481528a69b31f3398e650c09e","last_reissued_at":"2026-07-05T08:39:35.392109Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:39:35.392109Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMs' Classification Performance is Overclaimed","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Elizabeth Garrison, Elmira Talebianaraki, Hanzi Xu, Jiangshu Du, Renze Lou, Slobodan Vucetic, Vahid Mahzoon, Wenpeng Yin, Zhuoan Zhou","submitted_at":"2024-06-23T19:49:10Z","abstract_excerpt":"In many classification tasks designed for AI or human to solve, gold labels are typically included within the label space by default, often posed as \"which of the following is correct?\" This standard setup has traditionally highlighted the strong performance of advanced AI, particularly top-performing Large Language Models (LLMs), in routine classification tasks. However, when the gold label is intentionally excluded from the label space, it becomes evident that LLMs still attempt to select from the available label candidates, even when none are correct. This raises a pivotal question: Do LLMs"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16203","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.16203/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.16203","created_at":"2026-07-05T08:39:35.392164+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.16203v3","created_at":"2026-07-05T08:39:35.392164+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16203","created_at":"2026-07-05T08:39:35.392164+00:00"},{"alias_kind":"pith_short_12","alias_value":"FPL5SNEUQEJT","created_at":"2026-07-05T08:39:35.392164+00:00"},{"alias_kind":"pith_short_16","alias_value":"FPL5SNEUQEJT7EWR","created_at":"2026-07-05T08:39:35.392164+00:00"},{"alias_kind":"pith_short_8","alias_value":"FPL5SNEU","created_at":"2026-07-05T08:39:35.392164+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.21398","citing_title":"In a Few Words: Comparing Weak Supervision and LLMs for Short Query Intent Classification","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FPL5SNEUQEJT7EWRWGW6UIRPDW","json":"https://pith.science/pith/FPL5SNEUQEJT7EWRWGW6UIRPDW.json","graph_json":"https://pith.science/api/pith-number/FPL5SNEUQEJT7EWRWGW6UIRPDW/graph.json","events_json":"https://pith.science/api/pith-number/FPL5SNEUQEJT7EWRWGW6UIRPDW/events.json","paper":"https://pith.science/paper/FPL5SNEU"},"agent_actions":{"view_html":"https://pith.science/pith/FPL5SNEUQEJT7EWRWGW6UIRPDW","download_json":"https://pith.science/pith/FPL5SNEUQEJT7EWRWGW6UIRPDW.json","view_paper":"https://pith.science/paper/FPL5SNEU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.16203&json=true","fetch_graph":"https://pith.science/api/pith-number/FPL5SNEUQEJT7EWRWGW6UIRPDW/graph.json","fetch_events":"https://pith.science/api/pith-number/FPL5SNEUQEJT7EWRWGW6UIRPDW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FPL5SNEUQEJT7EWRWGW6UIRPDW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FPL5SNEUQEJT7EWRWGW6UIRPDW/action/storage_attestation","attest_author":"https://pith.science/pith/FPL5SNEUQEJT7EWRWGW6UIRPDW/action/author_attestation","sign_citation":"https://pith.science/pith/FPL5SNEUQEJT7EWRWGW6UIRPDW/action/citation_signature","submit_replication":"https://pith.science/pith/FPL5SNEUQEJT7EWRWGW6UIRPDW/action/replication_record"}},"created_at":"2026-07-05T08:39:35.392164+00:00","updated_at":"2026-07-05T08:39:35.392164+00:00"}