{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WBFBQRG3OI5SQMQ436IIGN4SLM","short_pith_number":"pith:WBFBQRG3","schema_version":"1.0","canonical_sha256":"b04a1844db723b28321cdf908337925b080fda6cab32346f456aef4a518bf62e","source":{"kind":"arxiv","id":"2401.05940","version":1},"attestation_state":"computed","paper":{"title":"Mutation-based Consistency Testing for Evaluating the Code Understanding Capability of LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Donghwan Shin, Ziyu Li","submitted_at":"2024-01-11T14:27:43Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable capabilities in processing both natural and programming languages, which have enabled various applications in software engineering, such as requirement engineering, code generation, and software testing. However, existing code generation benchmarks do not necessarily assess the code understanding performance of LLMs, especially for the subtle inconsistencies that may arise between code and its semantics described in natural language.\n  In this paper, we propose a novel method to systematically assess the code understanding performance of LLMs,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.05940","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-01-11T14:27:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"893134f4f8797d00a0ddd76f52c00d2c8ac6112faa4cca9f158e7e09dc88ae89","abstract_canon_sha256":"5ab6f731eb5e5deff213d4090255f65ad8cc2229d3c58829df89600047ab3360"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:32:38.201658Z","signature_b64":"TcSKRKtfAVd5ZuA2nUajHs4l30VLZitfA1QmVBezIhVIZ44nwjevhld2lXhpnOrK1zX+iwgHj3TK9VtgMX7fCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b04a1844db723b28321cdf908337925b080fda6cab32346f456aef4a518bf62e","last_reissued_at":"2026-07-05T07:32:38.201127Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:32:38.201127Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mutation-based Consistency Testing for Evaluating the Code Understanding Capability of LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Donghwan Shin, Ziyu Li","submitted_at":"2024-01-11T14:27:43Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable capabilities in processing both natural and programming languages, which have enabled various applications in software engineering, such as requirement engineering, code generation, and software testing. However, existing code generation benchmarks do not necessarily assess the code understanding performance of LLMs, especially for the subtle inconsistencies that may arise between code and its semantics described in natural language.\n  In this paper, we propose a novel method to systematically assess the code understanding performance of LLMs,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.05940","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.05940/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.05940","created_at":"2026-07-05T07:32:38.201185+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.05940v1","created_at":"2026-07-05T07:32:38.201185+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.05940","created_at":"2026-07-05T07:32:38.201185+00:00"},{"alias_kind":"pith_short_12","alias_value":"WBFBQRG3OI5S","created_at":"2026-07-05T07:32:38.201185+00:00"},{"alias_kind":"pith_short_16","alias_value":"WBFBQRG3OI5SQMQ4","created_at":"2026-07-05T07:32:38.201185+00:00"},{"alias_kind":"pith_short_8","alias_value":"WBFBQRG3","created_at":"2026-07-05T07:32:38.201185+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.02794","citing_title":"METAMON: Finding Inconsistencies between Program Documentation and Behavior using Metamorphic LLM Queries","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WBFBQRG3OI5SQMQ436IIGN4SLM","json":"https://pith.science/pith/WBFBQRG3OI5SQMQ436IIGN4SLM.json","graph_json":"https://pith.science/api/pith-number/WBFBQRG3OI5SQMQ436IIGN4SLM/graph.json","events_json":"https://pith.science/api/pith-number/WBFBQRG3OI5SQMQ436IIGN4SLM/events.json","paper":"https://pith.science/paper/WBFBQRG3"},"agent_actions":{"view_html":"https://pith.science/pith/WBFBQRG3OI5SQMQ436IIGN4SLM","download_json":"https://pith.science/pith/WBFBQRG3OI5SQMQ436IIGN4SLM.json","view_paper":"https://pith.science/paper/WBFBQRG3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.05940&json=true","fetch_graph":"https://pith.science/api/pith-number/WBFBQRG3OI5SQMQ436IIGN4SLM/graph.json","fetch_events":"https://pith.science/api/pith-number/WBFBQRG3OI5SQMQ436IIGN4SLM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WBFBQRG3OI5SQMQ436IIGN4SLM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WBFBQRG3OI5SQMQ436IIGN4SLM/action/storage_attestation","attest_author":"https://pith.science/pith/WBFBQRG3OI5SQMQ436IIGN4SLM/action/author_attestation","sign_citation":"https://pith.science/pith/WBFBQRG3OI5SQMQ436IIGN4SLM/action/citation_signature","submit_replication":"https://pith.science/pith/WBFBQRG3OI5SQMQ436IIGN4SLM/action/replication_record"}},"created_at":"2026-07-05T07:32:38.201185+00:00","updated_at":"2026-07-05T07:32:38.201185+00:00"}