{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZLU4ABLQNH7OORCPJP7IQT5RFE","short_pith_number":"pith:ZLU4ABLQ","schema_version":"1.0","canonical_sha256":"cae9c0057069fee7444f4bfe884fb12933a6582ac836e95ff1eff903cb85bb90","source":{"kind":"arxiv","id":"2408.08978","version":2},"attestation_state":"computed","paper":{"title":"See What LLMs Cannot Answer: A Self-Challenge Framework for Uncovering LLM Weaknesses","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenguang Zhu, Jianhao Yan, Ming Zhong, Xuefeng Bai, Yang Liu, Yinghao Yang, Yue Zhang, Yulong Chen, Ziyi Yang","submitted_at":"2024-08-16T19:01:52Z","abstract_excerpt":"The impressive performance of Large Language Models (LLMs) has consistently surpassed numerous human-designed benchmarks, presenting new challenges in assessing the shortcomings of LLMs. Designing tasks and finding LLMs' limitations are becoming increasingly important. In this paper, we investigate the question of whether an LLM can discover its own limitations from the errors it makes. To this end, we propose a Self-Challenge evaluation framework with human-in-the-loop. Starting from seed instances that GPT-4 fails to answer, we prompt GPT-4 to summarize error patterns that can be used to gen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.08978","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-16T19:01:52Z","cross_cats_sorted":[],"title_canon_sha256":"fb63b020f5a8bd4f91c35638c55e9185a7234141007a1e80c7ff15103f5701c5","abstract_canon_sha256":"7dc7ecee95f3566bec1a36f2102c003cdbd6b4d50011410c577f2f7cea337822"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:55.929824Z","signature_b64":"RZHtkbklNG+9PWjQh03X+hDMzwyqnuY3ZUkpR0/4wKsDrr1gZfjP3qSoF3w4StNlDPWk2epgnsZG5iFJL3okBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cae9c0057069fee7444f4bfe884fb12933a6582ac836e95ff1eff903cb85bb90","last_reissued_at":"2026-07-05T09:13:55.929371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:55.929371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"See What LLMs Cannot Answer: A Self-Challenge Framework for Uncovering LLM Weaknesses","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenguang Zhu, Jianhao Yan, Ming Zhong, Xuefeng Bai, Yang Liu, Yinghao Yang, Yue Zhang, Yulong Chen, Ziyi Yang","submitted_at":"2024-08-16T19:01:52Z","abstract_excerpt":"The impressive performance of Large Language Models (LLMs) has consistently surpassed numerous human-designed benchmarks, presenting new challenges in assessing the shortcomings of LLMs. Designing tasks and finding LLMs' limitations are becoming increasingly important. In this paper, we investigate the question of whether an LLM can discover its own limitations from the errors it makes. To this end, we propose a Self-Challenge evaluation framework with human-in-the-loop. Starting from seed instances that GPT-4 fails to answer, we prompt GPT-4 to summarize error patterns that can be used to gen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.08978","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.08978/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.08978","created_at":"2026-07-05T09:13:55.929429+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.08978v2","created_at":"2026-07-05T09:13:55.929429+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.08978","created_at":"2026-07-05T09:13:55.929429+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZLU4ABLQNH7O","created_at":"2026-07-05T09:13:55.929429+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZLU4ABLQNH7OORCP","created_at":"2026-07-05T09:13:55.929429+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZLU4ABLQ","created_at":"2026-07-05T09:13:55.929429+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.00425","citing_title":"The Gold Medals in an Empty Room: Diagnosing Metalinguistic Reasoning in LLMs with Camlang","ref_index":2021,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZLU4ABLQNH7OORCPJP7IQT5RFE","json":"https://pith.science/pith/ZLU4ABLQNH7OORCPJP7IQT5RFE.json","graph_json":"https://pith.science/api/pith-number/ZLU4ABLQNH7OORCPJP7IQT5RFE/graph.json","events_json":"https://pith.science/api/pith-number/ZLU4ABLQNH7OORCPJP7IQT5RFE/events.json","paper":"https://pith.science/paper/ZLU4ABLQ"},"agent_actions":{"view_html":"https://pith.science/pith/ZLU4ABLQNH7OORCPJP7IQT5RFE","download_json":"https://pith.science/pith/ZLU4ABLQNH7OORCPJP7IQT5RFE.json","view_paper":"https://pith.science/paper/ZLU4ABLQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.08978&json=true","fetch_graph":"https://pith.science/api/pith-number/ZLU4ABLQNH7OORCPJP7IQT5RFE/graph.json","fetch_events":"https://pith.science/api/pith-number/ZLU4ABLQNH7OORCPJP7IQT5RFE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZLU4ABLQNH7OORCPJP7IQT5RFE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZLU4ABLQNH7OORCPJP7IQT5RFE/action/storage_attestation","attest_author":"https://pith.science/pith/ZLU4ABLQNH7OORCPJP7IQT5RFE/action/author_attestation","sign_citation":"https://pith.science/pith/ZLU4ABLQNH7OORCPJP7IQT5RFE/action/citation_signature","submit_replication":"https://pith.science/pith/ZLU4ABLQNH7OORCPJP7IQT5RFE/action/replication_record"}},"created_at":"2026-07-05T09:13:55.929429+00:00","updated_at":"2026-07-05T09:13:55.929429+00:00"}