{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:TGHOI4DET6VZYS7HQBSLSDV3GR","short_pith_number":"pith:TGHOI4DE","schema_version":"1.0","canonical_sha256":"998ee470649fab9c4be78064b90ebb344808c14ed76b644f881a0f39b46e2bc1","source":{"kind":"arxiv","id":"2601.20147","version":3},"attestation_state":"computed","paper":{"title":"Not All Tokens Matter: Data-Centric Optimization for Efficient Code Summarization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Alexander Serebrenik, Antonio Mastropaolo, Massimiliano Di Penta, Saima Afrin, Tushar Sharma, Zaiyu Cheng","submitted_at":"2026-01-28T00:45:28Z","abstract_excerpt":"The rapid advancement of Large Language Models (LLMs) has revolutionized software engineering automation, particularly in automated code summarization, which enhances program comprehension and supports development activities. However, training LLMs for code summarization remains computationally expensive, with performance deteriorating on longer inputs-challenges that intensify when handling millions of code-comment pairs. We investigate strategic data optimization through targeted token reduction to minimize computational overhead while maintaining summary quality. We compare three token-leve"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2601.20147","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2026-01-28T00:45:28Z","cross_cats_sorted":[],"title_canon_sha256":"242f7295d6c336444cb1d567aa39a3b0f2e7d6be5deeab2fe6bd3f0a585741ed","abstract_canon_sha256":"357c8135686431ba2e38587920039a6628f6952f6dd21ab9477912d5b6ed0262"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-20T01:18:26.487971Z","signature_b64":"WwvT0LhidCm4ISng4Pmk4bhEbXvrB7xNHwOylti4okiE8mvR3ckcaQr7TzZ87QbxFL4Q0aJcIFAaCm9IQZ5BAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"998ee470649fab9c4be78064b90ebb344808c14ed76b644f881a0f39b46e2bc1","last_reissued_at":"2026-07-20T01:18:26.487063Z","signature_status":"signed_v1","first_computed_at":"2026-07-20T01:18:26.487063Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Not All Tokens Matter: Data-Centric Optimization for Efficient Code Summarization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Alexander Serebrenik, Antonio Mastropaolo, Massimiliano Di Penta, Saima Afrin, Tushar Sharma, Zaiyu Cheng","submitted_at":"2026-01-28T00:45:28Z","abstract_excerpt":"The rapid advancement of Large Language Models (LLMs) has revolutionized software engineering automation, particularly in automated code summarization, which enhances program comprehension and supports development activities. However, training LLMs for code summarization remains computationally expensive, with performance deteriorating on longer inputs-challenges that intensify when handling millions of code-comment pairs. We investigate strategic data optimization through targeted token reduction to minimize computational overhead while maintaining summary quality. We compare three token-leve"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2601.20147","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2601.20147/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2601.20147","created_at":"2026-07-20T01:18:26.487493+00:00"},{"alias_kind":"arxiv_version","alias_value":"2601.20147v3","created_at":"2026-07-20T01:18:26.487493+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2601.20147","created_at":"2026-07-20T01:18:26.487493+00:00"},{"alias_kind":"pith_short_12","alias_value":"TGHOI4DET6VZ","created_at":"2026-07-20T01:18:26.487493+00:00"},{"alias_kind":"pith_short_16","alias_value":"TGHOI4DET6VZYS7H","created_at":"2026-07-20T01:18:26.487493+00:00"},{"alias_kind":"pith_short_8","alias_value":"TGHOI4DE","created_at":"2026-07-20T01:18:26.487493+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TGHOI4DET6VZYS7HQBSLSDV3GR","json":"https://pith.science/pith/TGHOI4DET6VZYS7HQBSLSDV3GR.json","graph_json":"https://pith.science/api/pith-number/TGHOI4DET6VZYS7HQBSLSDV3GR/graph.json","events_json":"https://pith.science/api/pith-number/TGHOI4DET6VZYS7HQBSLSDV3GR/events.json","paper":"https://pith.science/paper/TGHOI4DE"},"agent_actions":{"view_html":"https://pith.science/pith/TGHOI4DET6VZYS7HQBSLSDV3GR","download_json":"https://pith.science/pith/TGHOI4DET6VZYS7HQBSLSDV3GR.json","view_paper":"https://pith.science/paper/TGHOI4DE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2601.20147&json=true","fetch_graph":"https://pith.science/api/pith-number/TGHOI4DET6VZYS7HQBSLSDV3GR/graph.json","fetch_events":"https://pith.science/api/pith-number/TGHOI4DET6VZYS7HQBSLSDV3GR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TGHOI4DET6VZYS7HQBSLSDV3GR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TGHOI4DET6VZYS7HQBSLSDV3GR/action/storage_attestation","attest_author":"https://pith.science/pith/TGHOI4DET6VZYS7HQBSLSDV3GR/action/author_attestation","sign_citation":"https://pith.science/pith/TGHOI4DET6VZYS7HQBSLSDV3GR/action/citation_signature","submit_replication":"https://pith.science/pith/TGHOI4DET6VZYS7HQBSLSDV3GR/action/replication_record"}},"created_at":"2026-07-20T01:18:26.487493+00:00","updated_at":"2026-07-20T01:18:26.487493+00:00"}