{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZBGWQIP3DRQABJLSCTQJ4AAHBD","short_pith_number":"pith:ZBGWQIP3","schema_version":"1.0","canonical_sha256":"c84d6821fb1c6000a57214e09e000708c9ff5afcd5957f29ec60b54a0e11bbc1","source":{"kind":"arxiv","id":"2509.04013","version":1},"attestation_state":"computed","paper":{"title":"On Robustness and Reliability of Benchmark-Based Evaluation of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Kevin Roitero, Riccardo Lunardi, Stefano Mizzaro, Vincenzo Della Mea","submitted_at":"2025-09-04T08:43:27Z","abstract_excerpt":"Large Language Models (LLMs) effectiveness is usually evaluated by means of benchmarks such as MMLU, ARC-C, or HellaSwag, where questions are presented in their original wording, thus in a fixed, standardized format. However, real-world applications involve linguistic variability, requiring models to maintain their effectiveness across diverse rewordings of the same question or query. In this study, we systematically assess the robustness of LLMs to paraphrased benchmark questions and investigate whether benchmark-based evaluations provide a reliable measure of model capabilities. We systemati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.04013","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-04T08:43:27Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ce535126dc35356e8a8eb79420c9cc7a8f53331ac9967b5b9c096080d04ae5ac","abstract_canon_sha256":"890eaf30bc06d618bac2f32057ad394755297ba5aadc743d6c7af1b4c54ccbf2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:04:51.857067Z","signature_b64":"kkf1/vKO0t342NyVFIL4aU0G6ySZ6L8mbsA2OaEb+3LWuau7W2CCGJT1m2kbWkKWV0FDNkfBrMEPECmldP3VCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c84d6821fb1c6000a57214e09e000708c9ff5afcd5957f29ec60b54a0e11bbc1","last_reissued_at":"2026-07-05T12:04:51.856569Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:04:51.856569Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Robustness and Reliability of Benchmark-Based Evaluation of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Kevin Roitero, Riccardo Lunardi, Stefano Mizzaro, Vincenzo Della Mea","submitted_at":"2025-09-04T08:43:27Z","abstract_excerpt":"Large Language Models (LLMs) effectiveness is usually evaluated by means of benchmarks such as MMLU, ARC-C, or HellaSwag, where questions are presented in their original wording, thus in a fixed, standardized format. However, real-world applications involve linguistic variability, requiring models to maintain their effectiveness across diverse rewordings of the same question or query. In this study, we systematically assess the robustness of LLMs to paraphrased benchmark questions and investigate whether benchmark-based evaluations provide a reliable measure of model capabilities. We systemati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.04013","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.04013/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.04013","created_at":"2026-07-05T12:04:51.856627+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.04013v1","created_at":"2026-07-05T12:04:51.856627+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.04013","created_at":"2026-07-05T12:04:51.856627+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZBGWQIP3DRQA","created_at":"2026-07-05T12:04:51.856627+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZBGWQIP3DRQABJLS","created_at":"2026-07-05T12:04:51.856627+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZBGWQIP3","created_at":"2026-07-05T12:04:51.856627+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.07053","citing_title":"GSM-SEM: Benchmark and Framework for Generating Semantically Variant Augmentations","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27209","citing_title":"Learning to Act under Noise: Enhancing Agent Robustness via Noisy Environments","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30738","citing_title":"MAVEN: Improving Generalization in Agentic Tool Calling","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19382","citing_title":"PRISM: A Benchmark for Programmatic Spatial-Temporal Reasoning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10687","citing_title":"Safe for Whom? Rethinking How We Evaluate the Safety of LLMs for Real Users","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26500","citing_title":"StarDrinks: An English and Korean Test Set for SLU Evaluation in a Drink Ordering Scenario","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07053","citing_title":"GSM-SEM: Benchmark and Framework for Generating Semantically Variant Augmentations","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZBGWQIP3DRQABJLSCTQJ4AAHBD","json":"https://pith.science/pith/ZBGWQIP3DRQABJLSCTQJ4AAHBD.json","graph_json":"https://pith.science/api/pith-number/ZBGWQIP3DRQABJLSCTQJ4AAHBD/graph.json","events_json":"https://pith.science/api/pith-number/ZBGWQIP3DRQABJLSCTQJ4AAHBD/events.json","paper":"https://pith.science/paper/ZBGWQIP3"},"agent_actions":{"view_html":"https://pith.science/pith/ZBGWQIP3DRQABJLSCTQJ4AAHBD","download_json":"https://pith.science/pith/ZBGWQIP3DRQABJLSCTQJ4AAHBD.json","view_paper":"https://pith.science/paper/ZBGWQIP3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.04013&json=true","fetch_graph":"https://pith.science/api/pith-number/ZBGWQIP3DRQABJLSCTQJ4AAHBD/graph.json","fetch_events":"https://pith.science/api/pith-number/ZBGWQIP3DRQABJLSCTQJ4AAHBD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZBGWQIP3DRQABJLSCTQJ4AAHBD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZBGWQIP3DRQABJLSCTQJ4AAHBD/action/storage_attestation","attest_author":"https://pith.science/pith/ZBGWQIP3DRQABJLSCTQJ4AAHBD/action/author_attestation","sign_citation":"https://pith.science/pith/ZBGWQIP3DRQABJLSCTQJ4AAHBD/action/citation_signature","submit_replication":"https://pith.science/pith/ZBGWQIP3DRQABJLSCTQJ4AAHBD/action/replication_record"}},"created_at":"2026-07-05T12:04:51.856627+00:00","updated_at":"2026-07-05T12:04:51.856627+00:00"}