{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JCFDNISWHQRBOHNI3XPUDXLDXD","short_pith_number":"pith:JCFDNISW","schema_version":"1.0","canonical_sha256":"488a36a2563c22171da8dddf41dd63b8dc48018476c89e0cc30ad4f038eb3843","source":{"kind":"arxiv","id":"2405.19740","version":2},"attestation_state":"computed","paper":{"title":"PertEval: Unveiling Real Knowledge Capacity of LLMs with Knowledge-Invariant Perturbations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Jiatong Li, Kunzhe Huang, Mengxiao Zhu, Qi Liu, Renjun Hu, Wei Lin, Xing Shi, Yan Zhuang","submitted_at":"2024-05-30T06:38:32Z","abstract_excerpt":"Expert-designed close-ended benchmarks are indispensable in assessing the knowledge capacity of large language models (LLMs). Despite their widespread use, concerns have mounted regarding their reliability due to limited test scenarios and an unavoidable risk of data contamination. To rectify this, we present PertEval, a toolkit devised for in-depth probing of LLMs' knowledge capacity through \\textbf{knowledge-invariant perturbations}. These perturbations employ human-like restatement techniques to generate on-the-fly test samples from static benchmarks, meticulously retaining knowledge-critic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.19740","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-30T06:38:32Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"92ecbd7fb16027edf309db4f19133e80d8e001aba9f9b091ec15d874b676d35e","abstract_canon_sha256":"5475cc402f8af777e2794fc432cb0ff30219d7d7ffad55b5e574d12aceade6fe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:22:18.594702Z","signature_b64":"L7UttfI+yfhxFvWi0wyE5IajQ5H6CMYg1CbGbygHWmE7ec3oa0DRMxdNO+G3UtyWJcwu6rCOBUNersaAmWm9BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"488a36a2563c22171da8dddf41dd63b8dc48018476c89e0cc30ad4f038eb3843","last_reissued_at":"2026-07-05T09:22:18.594215Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:22:18.594215Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PertEval: Unveiling Real Knowledge Capacity of LLMs with Knowledge-Invariant Perturbations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Jiatong Li, Kunzhe Huang, Mengxiao Zhu, Qi Liu, Renjun Hu, Wei Lin, Xing Shi, Yan Zhuang","submitted_at":"2024-05-30T06:38:32Z","abstract_excerpt":"Expert-designed close-ended benchmarks are indispensable in assessing the knowledge capacity of large language models (LLMs). Despite their widespread use, concerns have mounted regarding their reliability due to limited test scenarios and an unavoidable risk of data contamination. To rectify this, we present PertEval, a toolkit devised for in-depth probing of LLMs' knowledge capacity through \\textbf{knowledge-invariant perturbations}. These perturbations employ human-like restatement techniques to generate on-the-fly test samples from static benchmarks, meticulously retaining knowledge-critic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.19740","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.19740/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.19740","created_at":"2026-07-05T09:22:18.594271+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.19740v2","created_at":"2026-07-05T09:22:18.594271+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.19740","created_at":"2026-07-05T09:22:18.594271+00:00"},{"alias_kind":"pith_short_12","alias_value":"JCFDNISWHQRB","created_at":"2026-07-05T09:22:18.594271+00:00"},{"alias_kind":"pith_short_16","alias_value":"JCFDNISWHQRBOHNI","created_at":"2026-07-05T09:22:18.594271+00:00"},{"alias_kind":"pith_short_8","alias_value":"JCFDNISW","created_at":"2026-07-05T09:22:18.594271+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03606","citing_title":"Testing LLM Arithmetic Reasoning Generalization with Automatic Numeric-Remapping Attacks","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08614","citing_title":"DiagnosticIQ: A Benchmark for LLM-Based Industrial Maintenance Action Recommendation from Symbolic Rules","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JCFDNISWHQRBOHNI3XPUDXLDXD","json":"https://pith.science/pith/JCFDNISWHQRBOHNI3XPUDXLDXD.json","graph_json":"https://pith.science/api/pith-number/JCFDNISWHQRBOHNI3XPUDXLDXD/graph.json","events_json":"https://pith.science/api/pith-number/JCFDNISWHQRBOHNI3XPUDXLDXD/events.json","paper":"https://pith.science/paper/JCFDNISW"},"agent_actions":{"view_html":"https://pith.science/pith/JCFDNISWHQRBOHNI3XPUDXLDXD","download_json":"https://pith.science/pith/JCFDNISWHQRBOHNI3XPUDXLDXD.json","view_paper":"https://pith.science/paper/JCFDNISW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.19740&json=true","fetch_graph":"https://pith.science/api/pith-number/JCFDNISWHQRBOHNI3XPUDXLDXD/graph.json","fetch_events":"https://pith.science/api/pith-number/JCFDNISWHQRBOHNI3XPUDXLDXD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JCFDNISWHQRBOHNI3XPUDXLDXD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JCFDNISWHQRBOHNI3XPUDXLDXD/action/storage_attestation","attest_author":"https://pith.science/pith/JCFDNISWHQRBOHNI3XPUDXLDXD/action/author_attestation","sign_citation":"https://pith.science/pith/JCFDNISWHQRBOHNI3XPUDXLDXD/action/citation_signature","submit_replication":"https://pith.science/pith/JCFDNISWHQRBOHNI3XPUDXLDXD/action/replication_record"}},"created_at":"2026-07-05T09:22:18.594271+00:00","updated_at":"2026-07-05T09:22:18.594271+00:00"}