{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SLYORKRLGNK4CIXMQPLIIWVFZK","short_pith_number":"pith:SLYORKRL","schema_version":"1.0","canonical_sha256":"92f0e8aa2b3355c122ec83d6845aa5ca87406a3399c2ad4805a23765b9e44f9b","source":{"kind":"arxiv","id":"2501.03783","version":1},"attestation_state":"computed","paper":{"title":"How to Select Pre-Trained Code Models for Reuse? A Learning Perspective","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Guandong Xu, Hai Jin, Hongyu Zhang, Junyi Zhang, Yao Wan, Yufei Hu, Zhangqian Bi, Zhaoyang Chu","submitted_at":"2025-01-07T13:45:24Z","abstract_excerpt":"Pre-training a language model and then fine-tuning it has shown to be an efficient and effective technique for a wide range of code intelligence tasks, such as code generation, code summarization, and vulnerability detection. However, pretraining language models on a large-scale code corpus is computationally expensive. Fortunately, many off-the-shelf Pre-trained Code Models (PCMs), such as CodeBERT, CodeT5, CodeGen, and Code Llama, have been released publicly. These models acquire general code understanding and generation capability during pretraining, which enhances their performance on down"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.03783","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.SE","submitted_at":"2025-01-07T13:45:24Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8a15fdfb9700db15927966ab21d3f25b1473ff29614b35ead62e69a3542e3d93","abstract_canon_sha256":"058775e52865adfe5d0071ed368329b32325de9f76fff841ca98309ef99ad2de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:57:59.501960Z","signature_b64":"ufKL5s9+apN8KPeEHubLT5t8zlr1yA/XZvqOv33aXlfOD2sBrUAAZyrbNsjsQ9NNBGKkybJx2/VeVV9JAxmJDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"92f0e8aa2b3355c122ec83d6845aa5ca87406a3399c2ad4805a23765b9e44f9b","last_reissued_at":"2026-07-05T09:57:59.501423Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:57:59.501423Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How to Select Pre-Trained Code Models for Reuse? A Learning Perspective","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Guandong Xu, Hai Jin, Hongyu Zhang, Junyi Zhang, Yao Wan, Yufei Hu, Zhangqian Bi, Zhaoyang Chu","submitted_at":"2025-01-07T13:45:24Z","abstract_excerpt":"Pre-training a language model and then fine-tuning it has shown to be an efficient and effective technique for a wide range of code intelligence tasks, such as code generation, code summarization, and vulnerability detection. However, pretraining language models on a large-scale code corpus is computationally expensive. Fortunately, many off-the-shelf Pre-trained Code Models (PCMs), such as CodeBERT, CodeT5, CodeGen, and Code Llama, have been released publicly. These models acquire general code understanding and generation capability during pretraining, which enhances their performance on down"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.03783","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.03783/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.03783","created_at":"2026-07-05T09:57:59.501524+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.03783v1","created_at":"2026-07-05T09:57:59.501524+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.03783","created_at":"2026-07-05T09:57:59.501524+00:00"},{"alias_kind":"pith_short_12","alias_value":"SLYORKRLGNK4","created_at":"2026-07-05T09:57:59.501524+00:00"},{"alias_kind":"pith_short_16","alias_value":"SLYORKRLGNK4CIXM","created_at":"2026-07-05T09:57:59.501524+00:00"},{"alias_kind":"pith_short_8","alias_value":"SLYORKRL","created_at":"2026-07-05T09:57:59.501524+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SLYORKRLGNK4CIXMQPLIIWVFZK","json":"https://pith.science/pith/SLYORKRLGNK4CIXMQPLIIWVFZK.json","graph_json":"https://pith.science/api/pith-number/SLYORKRLGNK4CIXMQPLIIWVFZK/graph.json","events_json":"https://pith.science/api/pith-number/SLYORKRLGNK4CIXMQPLIIWVFZK/events.json","paper":"https://pith.science/paper/SLYORKRL"},"agent_actions":{"view_html":"https://pith.science/pith/SLYORKRLGNK4CIXMQPLIIWVFZK","download_json":"https://pith.science/pith/SLYORKRLGNK4CIXMQPLIIWVFZK.json","view_paper":"https://pith.science/paper/SLYORKRL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.03783&json=true","fetch_graph":"https://pith.science/api/pith-number/SLYORKRLGNK4CIXMQPLIIWVFZK/graph.json","fetch_events":"https://pith.science/api/pith-number/SLYORKRLGNK4CIXMQPLIIWVFZK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SLYORKRLGNK4CIXMQPLIIWVFZK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SLYORKRLGNK4CIXMQPLIIWVFZK/action/storage_attestation","attest_author":"https://pith.science/pith/SLYORKRLGNK4CIXMQPLIIWVFZK/action/author_attestation","sign_citation":"https://pith.science/pith/SLYORKRLGNK4CIXMQPLIIWVFZK/action/citation_signature","submit_replication":"https://pith.science/pith/SLYORKRLGNK4CIXMQPLIIWVFZK/action/replication_record"}},"created_at":"2026-07-05T09:57:59.501524+00:00","updated_at":"2026-07-05T09:57:59.501524+00:00"}