{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:YSZUZ4VBSQ4VI5MR7ZWZO4VCSY","short_pith_number":"pith:YSZUZ4VB","schema_version":"1.0","canonical_sha256":"c4b34cf2a19439547591fe6d9772a29609b5b58b7761c1792fef90dfdeec81f6","source":{"kind":"arxiv","id":"2602.17838","version":2},"attestation_state":"computed","paper":{"title":"Using Mutation-Analysis to Examine an LLM's Ability to Summarize Code","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Bogdan Vasilescu, Lara Khatib, Meiyappan Nagappan, Michael Pu","submitted_at":"2026-02-19T21:00:49Z","abstract_excerpt":"As developers increasingly rely on LLM-generated code summaries for documentation, testing, and review, it is important to study whether these summaries accurately reflect what the program actually does. LLMs often produce confident descriptions of what the code looks like it should do (intent), while missing subtle edge cases or logic changes that define what it actually does (behavior). We present a mutation-based evaluation methodology that directly tests whether a summary truly matches the code's logic. Our approach generates a summary, injects a targeted mutation into the code, and checks"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.17838","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2026-02-19T21:00:49Z","cross_cats_sorted":[],"title_canon_sha256":"81026f2695f1d9168a3af638e069a0934e59ac30cfc5c75e0a796f60a9526da6","abstract_canon_sha256":"53fed97d79d77f2e42b3581fe8f8dda42e9ab0ddacd8a783bcf2e62305151158"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-29T00:14:02.196657Z","signature_b64":"7iLF0/yK1bzB+Xe9v0I0bmR+cvJh96He77MGxOEEHXukfyUk/hzze4lh9JhBNcpyKJ5fu0bEy5NlO2RPgnIHBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4b34cf2a19439547591fe6d9772a29609b5b58b7761c1792fef90dfdeec81f6","last_reissued_at":"2026-06-29T00:14:02.196099Z","signature_status":"signed_v1","first_computed_at":"2026-06-29T00:14:02.196099Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Using Mutation-Analysis to Examine an LLM's Ability to Summarize Code","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Bogdan Vasilescu, Lara Khatib, Meiyappan Nagappan, Michael Pu","submitted_at":"2026-02-19T21:00:49Z","abstract_excerpt":"As developers increasingly rely on LLM-generated code summaries for documentation, testing, and review, it is important to study whether these summaries accurately reflect what the program actually does. LLMs often produce confident descriptions of what the code looks like it should do (intent), while missing subtle edge cases or logic changes that define what it actually does (behavior). We present a mutation-based evaluation methodology that directly tests whether a summary truly matches the code's logic. Our approach generates a summary, injects a targeted mutation into the code, and checks"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.17838","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.17838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.17838","created_at":"2026-06-29T00:14:02.196161+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.17838v2","created_at":"2026-06-29T00:14:02.196161+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.17838","created_at":"2026-06-29T00:14:02.196161+00:00"},{"alias_kind":"pith_short_12","alias_value":"YSZUZ4VBSQ4V","created_at":"2026-06-29T00:14:02.196161+00:00"},{"alias_kind":"pith_short_16","alias_value":"YSZUZ4VBSQ4VI5MR","created_at":"2026-06-29T00:14:02.196161+00:00"},{"alias_kind":"pith_short_8","alias_value":"YSZUZ4VB","created_at":"2026-06-29T00:14:02.196161+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2605.13898","citing_title":"Bidirectional Empowerment of Metamorphic Testing and Large Language Models: A Systematic Survey","ref_index":47,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY","json":"https://pith.science/pith/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY.json","graph_json":"https://pith.science/api/pith-number/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY/graph.json","events_json":"https://pith.science/api/pith-number/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY/events.json","paper":"https://pith.science/paper/YSZUZ4VB"},"agent_actions":{"view_html":"https://pith.science/pith/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY","download_json":"https://pith.science/pith/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY.json","view_paper":"https://pith.science/paper/YSZUZ4VB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.17838&json=true","fetch_graph":"https://pith.science/api/pith-number/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY/graph.json","fetch_events":"https://pith.science/api/pith-number/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY/action/storage_attestation","attest_author":"https://pith.science/pith/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY/action/author_attestation","sign_citation":"https://pith.science/pith/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY/action/citation_signature","submit_replication":"https://pith.science/pith/YSZUZ4VBSQ4VI5MR7ZWZO4VCSY/action/replication_record"}},"created_at":"2026-06-29T00:14:02.196161+00:00","updated_at":"2026-06-29T00:14:02.196161+00:00"}