{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RTCYJXSZ3E7XIAPZQFSU2P72QZ","short_pith_number":"pith:RTCYJXSZ","schema_version":"1.0","canonical_sha256":"8cc584de59d93f7401f981654d3ffa86511ff0744e7fba22dcdf718ff48de0c0","source":{"kind":"arxiv","id":"2305.19915","version":4},"attestation_state":"computed","paper":{"title":"Source Code Data Augmentation for Deep Learning: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.CL","authors_text":"David Lo, Li Li, Terry Yue Zhuo, Xiaoning Du, Yufei Wang, Zhenchang Xing, Zhensu Sun, Zhou Yang","submitted_at":"2023-05-31T14:47:44Z","abstract_excerpt":"The increasingly popular adoption of deep learning models in many critical source code tasks motivates the development of data augmentation (DA) techniques to enhance training data and improve various capabilities (e.g., robustness and generalizability) of these models. Although a series of DA methods have been proposed and tailored for source code models, there lacks a comprehensive survey and examination to understand their effectiveness and implications. This paper fills this gap by conducting a comprehensive and integrative survey of data augmentation for source code, wherein we systematic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.19915","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-31T14:47:44Z","cross_cats_sorted":["cs.AI","cs.SE"],"title_canon_sha256":"006f2f31dd0a7082a9418ecc35f4e1472f988568eb6a3961b90c752c3d1f58fe","abstract_canon_sha256":"5cb10009ecbc4e1f4fea90531780fac7e4cc51b9ccf9e675fa90c22fe76354a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:56.404524Z","signature_b64":"M7xQ4rrMapfqjsvTKsvUWklKbQNHDJ8Tlr/SljFXdSFX3ZiPhlNg3l874BTE4NacLIZO5A6rBn9qTE6vceWYBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8cc584de59d93f7401f981654d3ffa86511ff0744e7fba22dcdf718ff48de0c0","last_reissued_at":"2026-07-05T07:11:56.404033Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:56.404033Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Source Code Data Augmentation for Deep Learning: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.CL","authors_text":"David Lo, Li Li, Terry Yue Zhuo, Xiaoning Du, Yufei Wang, Zhenchang Xing, Zhensu Sun, Zhou Yang","submitted_at":"2023-05-31T14:47:44Z","abstract_excerpt":"The increasingly popular adoption of deep learning models in many critical source code tasks motivates the development of data augmentation (DA) techniques to enhance training data and improve various capabilities (e.g., robustness and generalizability) of these models. Although a series of DA methods have been proposed and tailored for source code models, there lacks a comprehensive survey and examination to understand their effectiveness and implications. This paper fills this gap by conducting a comprehensive and integrative survey of data augmentation for source code, wherein we systematic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.19915","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.19915/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.19915","created_at":"2026-07-05T07:11:56.404091+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.19915v4","created_at":"2026-07-05T07:11:56.404091+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.19915","created_at":"2026-07-05T07:11:56.404091+00:00"},{"alias_kind":"pith_short_12","alias_value":"RTCYJXSZ3E7X","created_at":"2026-07-05T07:11:56.404091+00:00"},{"alias_kind":"pith_short_16","alias_value":"RTCYJXSZ3E7XIAPZ","created_at":"2026-07-05T07:11:56.404091+00:00"},{"alias_kind":"pith_short_8","alias_value":"RTCYJXSZ","created_at":"2026-07-05T07:11:56.404091+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2402.19173","citing_title":"StarCoder 2 and The Stack v2: The Next Generation","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RTCYJXSZ3E7XIAPZQFSU2P72QZ","json":"https://pith.science/pith/RTCYJXSZ3E7XIAPZQFSU2P72QZ.json","graph_json":"https://pith.science/api/pith-number/RTCYJXSZ3E7XIAPZQFSU2P72QZ/graph.json","events_json":"https://pith.science/api/pith-number/RTCYJXSZ3E7XIAPZQFSU2P72QZ/events.json","paper":"https://pith.science/paper/RTCYJXSZ"},"agent_actions":{"view_html":"https://pith.science/pith/RTCYJXSZ3E7XIAPZQFSU2P72QZ","download_json":"https://pith.science/pith/RTCYJXSZ3E7XIAPZQFSU2P72QZ.json","view_paper":"https://pith.science/paper/RTCYJXSZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.19915&json=true","fetch_graph":"https://pith.science/api/pith-number/RTCYJXSZ3E7XIAPZQFSU2P72QZ/graph.json","fetch_events":"https://pith.science/api/pith-number/RTCYJXSZ3E7XIAPZQFSU2P72QZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RTCYJXSZ3E7XIAPZQFSU2P72QZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RTCYJXSZ3E7XIAPZQFSU2P72QZ/action/storage_attestation","attest_author":"https://pith.science/pith/RTCYJXSZ3E7XIAPZQFSU2P72QZ/action/author_attestation","sign_citation":"https://pith.science/pith/RTCYJXSZ3E7XIAPZQFSU2P72QZ/action/citation_signature","submit_replication":"https://pith.science/pith/RTCYJXSZ3E7XIAPZQFSU2P72QZ/action/replication_record"}},"created_at":"2026-07-05T07:11:56.404091+00:00","updated_at":"2026-07-05T07:11:56.404091+00:00"}