{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:7S2SEOVFXVKF42KRTUUVUGE35T","short_pith_number":"pith:7S2SEOVF","schema_version":"1.0","canonical_sha256":"fcb5223aa5bd545e69519d295a189becc6965676f316ef2fa2f8ee61ca9b53e4","source":{"kind":"arxiv","id":"2103.06333","version":2},"attestation_state":"computed","paper":{"title":"Unified Pre-training for Program Understanding and Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.PL"],"primary_cat":"cs.CL","authors_text":"Baishakhi Ray, Kai-Wei Chang, Saikat Chakraborty, Wasi Uddin Ahmad","submitted_at":"2021-03-10T20:32:59Z","abstract_excerpt":"Code summarization and generation empower conversion between programming language (PL) and natural language (NL), while code translation avails the migration of legacy code from one PL to another. This paper introduces PLBART, a sequence-to-sequence model capable of performing a broad spectrum of program and language understanding and generation tasks. PLBART is pre-trained on an extensive collection of Java and Python functions and associated NL text via denoising autoencoding. Experiments on code summarization in the English language, code generation, and code translation in seven programmin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.06333","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-03-10T20:32:59Z","cross_cats_sorted":["cs.PL"],"title_canon_sha256":"c048eeab4eb655dd6ea9e3da3618cbe91a3a82af2460351e9d6043f42580f446","abstract_canon_sha256":"2ebc88bd6c4f7197785a35059c5e0063773490ac0090579b9cd67d3a1b3b3bf8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:30:43.865002Z","signature_b64":"alOQmeJWVlrYW4GKpQDnojZ3gbamHa15m/KL6m1kfwQVKeqf6UY23ipA+wVUsyuqnTh/zblk6tNfiwoa2E2AAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fcb5223aa5bd545e69519d295a189becc6965676f316ef2fa2f8ee61ca9b53e4","last_reissued_at":"2026-07-05T02:30:43.864500Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:30:43.864500Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unified Pre-training for Program Understanding and Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.PL"],"primary_cat":"cs.CL","authors_text":"Baishakhi Ray, Kai-Wei Chang, Saikat Chakraborty, Wasi Uddin Ahmad","submitted_at":"2021-03-10T20:32:59Z","abstract_excerpt":"Code summarization and generation empower conversion between programming language (PL) and natural language (NL), while code translation avails the migration of legacy code from one PL to another. This paper introduces PLBART, a sequence-to-sequence model capable of performing a broad spectrum of program and language understanding and generation tasks. PLBART is pre-trained on an extensive collection of Java and Python functions and associated NL text via denoising autoencoding. Experiments on code summarization in the English language, code generation, and code translation in seven programmin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.06333","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.06333/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.06333","created_at":"2026-07-05T02:30:43.864552+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.06333v2","created_at":"2026-07-05T02:30:43.864552+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.06333","created_at":"2026-07-05T02:30:43.864552+00:00"},{"alias_kind":"pith_short_12","alias_value":"7S2SEOVFXVKF","created_at":"2026-07-05T02:30:43.864552+00:00"},{"alias_kind":"pith_short_16","alias_value":"7S2SEOVFXVKF42KR","created_at":"2026-07-05T02:30:43.864552+00:00"},{"alias_kind":"pith_short_8","alias_value":"7S2SEOVF","created_at":"2026-07-05T02:30:43.864552+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00511","citing_title":"Large Language Models for Multi-Lingual Equivalent Mutant Detection: An Extended Empirical Study","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2305.12138","citing_title":"Exploring Code Analysis: Zero-Shot Insights on Syntax and Semantics with LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2403.16032","citing_title":"DeepFWI: Identifying Bug-Sensitive Warnings with Multi-Modal Code-Warning Semantics","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2410.22240","citing_title":"Are Decoder-Only Large Language Models the Silver Bullet for Code Search?","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19102","citing_title":"Prompt Optimization for LLM Code Generation via Reinforcement Learning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2508.03949","citing_title":"Model Compression vs. Adversarial Robustness: An Empirical Study on Language Models for Code","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21816","citing_title":"From Task to Tutorial: An Automated GUI Framework for Excel Tutorial Document and Video Creation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04166","citing_title":"Multi Language Models for On-the-Fly Syntax Highlighting","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2511.05476","citing_title":"A Metamorphic Testing Perspective on Knowledge Distillation for Language Models of Code: Does the Student Deeply Mimic the Teacher?","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2312.13010","citing_title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2406.00515","citing_title":"A Survey on Large Language Models for Code Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25599","citing_title":"PLMGH: What Matters in PLM-GNN Hybrids for Code Classification and Vulnerability Detection","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13107","citing_title":"Can Coding Agents Be General Agents?","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02215","citing_title":"HEJ-Robust: A Robustness Benchmark for LLM-Based Automated Program Repair","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15385","citing_title":"Prompt-Driven Code Summarization: A Systematic Literature Review","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02215","citing_title":"HEJ-Robust: A Robustness Benchmark for LLM-Based Automated Program Repair","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7S2SEOVFXVKF42KRTUUVUGE35T","json":"https://pith.science/pith/7S2SEOVFXVKF42KRTUUVUGE35T.json","graph_json":"https://pith.science/api/pith-number/7S2SEOVFXVKF42KRTUUVUGE35T/graph.json","events_json":"https://pith.science/api/pith-number/7S2SEOVFXVKF42KRTUUVUGE35T/events.json","paper":"https://pith.science/paper/7S2SEOVF"},"agent_actions":{"view_html":"https://pith.science/pith/7S2SEOVFXVKF42KRTUUVUGE35T","download_json":"https://pith.science/pith/7S2SEOVFXVKF42KRTUUVUGE35T.json","view_paper":"https://pith.science/paper/7S2SEOVF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.06333&json=true","fetch_graph":"https://pith.science/api/pith-number/7S2SEOVFXVKF42KRTUUVUGE35T/graph.json","fetch_events":"https://pith.science/api/pith-number/7S2SEOVFXVKF42KRTUUVUGE35T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7S2SEOVFXVKF42KRTUUVUGE35T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7S2SEOVFXVKF42KRTUUVUGE35T/action/storage_attestation","attest_author":"https://pith.science/pith/7S2SEOVFXVKF42KRTUUVUGE35T/action/author_attestation","sign_citation":"https://pith.science/pith/7S2SEOVFXVKF42KRTUUVUGE35T/action/citation_signature","submit_replication":"https://pith.science/pith/7S2SEOVFXVKF42KRTUUVUGE35T/action/replication_record"}},"created_at":"2026-07-05T02:30:43.864552+00:00","updated_at":"2026-07-05T02:30:43.864552+00:00"}