{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZHFIV3444QWIXPGIKMEJDEESXF","short_pith_number":"pith:ZHFIV344","schema_version":"1.0","canonical_sha256":"c9ca8aef9ce42c8bbcc85308919092b94c764a8ea3fcfe9dfa6b0950ca2849f8","source":{"kind":"arxiv","id":"2409.14644","version":3},"attestation_state":"computed","paper":{"title":"An Effective Approach to Embedding Source Code by Combining Large Language and Sentence Embedding Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Chenhui Cui, Chunrong Fang, Rubing Huang, Zhenyu Chen, Zixiang Xian","submitted_at":"2024-09-23T01:03:15Z","abstract_excerpt":"The advent of large language models (LLMs) has significantly advanced artificial intelligence (AI) in software engineering (SE), with source code embeddings playing a crucial role in tasks such as source code clone detection and source code clustering. However, existing methods for source code embedding, including those based on LLMs, often rely on costly supervised training or fine-tuning for domain adaptation. This paper proposes a novel approach to embedding source code by combining large language and sentence embedding models. This approach attempts to eliminate the need for task-specific "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.14644","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2024-09-23T01:03:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3fa9e30e92573e0a89fcba8dadcd542bbfb1a766bf334586199acdbb4ede5e46","abstract_canon_sha256":"06032f3e3311303ce7c067da4b51403257e8104fc78a35f288c65f290d0f3bab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:49.201974Z","signature_b64":"0Zrt/OjDg4pUmwjRBCldNI8bCXrxtO2yd1fQ97dKmesk0jvCge0+Quhawrh+VLYNNpwi1caNJ4cZGSElLICjCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9ca8aef9ce42c8bbcc85308919092b94c764a8ea3fcfe9dfa6b0950ca2849f8","last_reissued_at":"2026-07-05T11:14:49.201541Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:49.201541Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Effective Approach to Embedding Source Code by Combining Large Language and Sentence Embedding Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Chenhui Cui, Chunrong Fang, Rubing Huang, Zhenyu Chen, Zixiang Xian","submitted_at":"2024-09-23T01:03:15Z","abstract_excerpt":"The advent of large language models (LLMs) has significantly advanced artificial intelligence (AI) in software engineering (SE), with source code embeddings playing a crucial role in tasks such as source code clone detection and source code clustering. However, existing methods for source code embedding, including those based on LLMs, often rely on costly supervised training or fine-tuning for domain adaptation. This paper proposes a novel approach to embedding source code by combining large language and sentence embedding models. This approach attempts to eliminate the need for task-specific "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.14644","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.14644/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.14644","created_at":"2026-07-05T11:14:49.201595+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.14644v3","created_at":"2026-07-05T11:14:49.201595+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.14644","created_at":"2026-07-05T11:14:49.201595+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZHFIV3444QWI","created_at":"2026-07-05T11:14:49.201595+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZHFIV3444QWIXPGI","created_at":"2026-07-05T11:14:49.201595+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZHFIV344","created_at":"2026-07-05T11:14:49.201595+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.02368","citing_title":"Evaluating the Effectiveness of LLMs in Fixing Maintainability Issues in Real-World Projects","ref_index":42,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZHFIV3444QWIXPGIKMEJDEESXF","json":"https://pith.science/pith/ZHFIV3444QWIXPGIKMEJDEESXF.json","graph_json":"https://pith.science/api/pith-number/ZHFIV3444QWIXPGIKMEJDEESXF/graph.json","events_json":"https://pith.science/api/pith-number/ZHFIV3444QWIXPGIKMEJDEESXF/events.json","paper":"https://pith.science/paper/ZHFIV344"},"agent_actions":{"view_html":"https://pith.science/pith/ZHFIV3444QWIXPGIKMEJDEESXF","download_json":"https://pith.science/pith/ZHFIV3444QWIXPGIKMEJDEESXF.json","view_paper":"https://pith.science/paper/ZHFIV344","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.14644&json=true","fetch_graph":"https://pith.science/api/pith-number/ZHFIV3444QWIXPGIKMEJDEESXF/graph.json","fetch_events":"https://pith.science/api/pith-number/ZHFIV3444QWIXPGIKMEJDEESXF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZHFIV3444QWIXPGIKMEJDEESXF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZHFIV3444QWIXPGIKMEJDEESXF/action/storage_attestation","attest_author":"https://pith.science/pith/ZHFIV3444QWIXPGIKMEJDEESXF/action/author_attestation","sign_citation":"https://pith.science/pith/ZHFIV3444QWIXPGIKMEJDEESXF/action/citation_signature","submit_replication":"https://pith.science/pith/ZHFIV3444QWIXPGIKMEJDEESXF/action/replication_record"}},"created_at":"2026-07-05T11:14:49.201595+00:00","updated_at":"2026-07-05T11:14:49.201595+00:00"}