{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QRRY4C2PJUUN4ODUAAZQYEY3WS","short_pith_number":"pith:QRRY4C2P","schema_version":"1.0","canonical_sha256":"84638e0b4f4d28de387400330c131bb48fb9aa49d04aaed255426f70d0575ad2","source":{"kind":"arxiv","id":"2312.15475","version":1},"attestation_state":"computed","paper":{"title":"Evaluating Code Summarization Techniques: A New Metric and an Empirical Characterization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Antonio Mastropaolo, Gabriele Bavota, Massimiliano Di Penta, Matteo Ciniselli","submitted_at":"2023-12-24T13:12:39Z","abstract_excerpt":"Several code summarization techniques have been proposed in the literature to automatically document a code snippet or a function. Ideally, software developers should be involved in assessing the quality of the generated summaries. However, in most cases, researchers rely on automatic evaluation metrics such as BLEU, ROUGE, and METEOR. These metrics are all based on the same assumption: The higher the textual similarity between the generated summary and a reference summary written by developers, the higher its quality. However, there are two reasons for which this assumption falls short: (i) r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.15475","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2023-12-24T13:12:39Z","cross_cats_sorted":[],"title_canon_sha256":"20482e3e6b6e4b2c655017c78b43a3425cc711b00e9a7c51e200c4c8dabc732f","abstract_canon_sha256":"26ca7b547371e5ee9d9e9f448c8d1de7ac31424b184a6d720aeb2cca908e3cc0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:27:57.085361Z","signature_b64":"mVpJ/xgy3Fw2enE5IMuS1Ex2ndKALhbdZLEzDggz+FPTUdddLfROuv7C9uwOWIYsbz/12fyVEfevk2krV1ZDDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"84638e0b4f4d28de387400330c131bb48fb9aa49d04aaed255426f70d0575ad2","last_reissued_at":"2026-07-05T07:27:57.084924Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:27:57.084924Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Code Summarization Techniques: A New Metric and an Empirical Characterization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Antonio Mastropaolo, Gabriele Bavota, Massimiliano Di Penta, Matteo Ciniselli","submitted_at":"2023-12-24T13:12:39Z","abstract_excerpt":"Several code summarization techniques have been proposed in the literature to automatically document a code snippet or a function. Ideally, software developers should be involved in assessing the quality of the generated summaries. However, in most cases, researchers rely on automatic evaluation metrics such as BLEU, ROUGE, and METEOR. These metrics are all based on the same assumption: The higher the textual similarity between the generated summary and a reference summary written by developers, the higher its quality. However, there are two reasons for which this assumption falls short: (i) r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.15475","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.15475/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.15475","created_at":"2026-07-05T07:27:57.084977+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.15475v1","created_at":"2026-07-05T07:27:57.084977+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.15475","created_at":"2026-07-05T07:27:57.084977+00:00"},{"alias_kind":"pith_short_12","alias_value":"QRRY4C2PJUUN","created_at":"2026-07-05T07:27:57.084977+00:00"},{"alias_kind":"pith_short_16","alias_value":"QRRY4C2PJUUN4ODU","created_at":"2026-07-05T07:27:57.084977+00:00"},{"alias_kind":"pith_short_8","alias_value":"QRRY4C2P","created_at":"2026-07-05T07:27:57.084977+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.13737","citing_title":"On the Compression of Language Models for Code: An Empirical Study on CodeBERT","ref_index":36,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QRRY4C2PJUUN4ODUAAZQYEY3WS","json":"https://pith.science/pith/QRRY4C2PJUUN4ODUAAZQYEY3WS.json","graph_json":"https://pith.science/api/pith-number/QRRY4C2PJUUN4ODUAAZQYEY3WS/graph.json","events_json":"https://pith.science/api/pith-number/QRRY4C2PJUUN4ODUAAZQYEY3WS/events.json","paper":"https://pith.science/paper/QRRY4C2P"},"agent_actions":{"view_html":"https://pith.science/pith/QRRY4C2PJUUN4ODUAAZQYEY3WS","download_json":"https://pith.science/pith/QRRY4C2PJUUN4ODUAAZQYEY3WS.json","view_paper":"https://pith.science/paper/QRRY4C2P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.15475&json=true","fetch_graph":"https://pith.science/api/pith-number/QRRY4C2PJUUN4ODUAAZQYEY3WS/graph.json","fetch_events":"https://pith.science/api/pith-number/QRRY4C2PJUUN4ODUAAZQYEY3WS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QRRY4C2PJUUN4ODUAAZQYEY3WS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QRRY4C2PJUUN4ODUAAZQYEY3WS/action/storage_attestation","attest_author":"https://pith.science/pith/QRRY4C2PJUUN4ODUAAZQYEY3WS/action/author_attestation","sign_citation":"https://pith.science/pith/QRRY4C2PJUUN4ODUAAZQYEY3WS/action/citation_signature","submit_replication":"https://pith.science/pith/QRRY4C2PJUUN4ODUAAZQYEY3WS/action/replication_record"}},"created_at":"2026-07-05T07:27:57.084977+00:00","updated_at":"2026-07-05T07:27:57.084977+00:00"}