{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QA4BKTABBOSJB3OBZRYE2T7IFW","short_pith_number":"pith:QA4BKTAB","schema_version":"1.0","canonical_sha256":"8038154c010ba490edc1cc704d4fe82d8dfa2b803a24416f5977a4912685e5d7","source":{"kind":"arxiv","id":"2306.10998","version":1},"attestation_state":"computed","paper":{"title":"RepoFusion: Training Code Models to Understand Your Repository","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.PL","cs.SE"],"primary_cat":"cs.LG","authors_text":"Denis Kocetkov, Disha Shrivastava, Dzmitry Bahdanau, Harm de Vries, Torsten Scholak","submitted_at":"2023-06-19T15:05:31Z","abstract_excerpt":"Despite the huge success of Large Language Models (LLMs) in coding assistants like GitHub Copilot, these models struggle to understand the context present in the repository (e.g., imports, parent classes, files with similar names, etc.), thereby producing inaccurate code completions. This effect is more pronounced when using these assistants for repositories that the model has not seen during training, such as proprietary software or work-in-progress code projects. Recent work has shown the promise of using context from the repository during inference. In this work, we extend this idea and pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.10998","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-19T15:05:31Z","cross_cats_sorted":["cs.AI","cs.PL","cs.SE"],"title_canon_sha256":"fae61b1bb6f4b9a253b318b0e5af960e3a912647278595561804f0e11d4b39db","abstract_canon_sha256":"7c69a708d528a9ee711a172e3bdc80a7ae6aedb5605eb1f754738a5a1d205092"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:22:14.969048Z","signature_b64":"uKKsXbw2MlsSb4bzijJS+TuOatV/OrT11I8vxFNVHpRwqM0EfJB2emu/Ri5/8MHbMgob5x6aRkx/rZ5Oz0dAAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8038154c010ba490edc1cc704d4fe82d8dfa2b803a24416f5977a4912685e5d7","last_reissued_at":"2026-07-05T06:22:14.968593Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:22:14.968593Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RepoFusion: Training Code Models to Understand Your Repository","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.PL","cs.SE"],"primary_cat":"cs.LG","authors_text":"Denis Kocetkov, Disha Shrivastava, Dzmitry Bahdanau, Harm de Vries, Torsten Scholak","submitted_at":"2023-06-19T15:05:31Z","abstract_excerpt":"Despite the huge success of Large Language Models (LLMs) in coding assistants like GitHub Copilot, these models struggle to understand the context present in the repository (e.g., imports, parent classes, files with similar names, etc.), thereby producing inaccurate code completions. This effect is more pronounced when using these assistants for repositories that the model has not seen during training, such as proprietary software or work-in-progress code projects. Recent work has shown the promise of using context from the repository during inference. In this work, we extend this idea and pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.10998","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.10998/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.10998","created_at":"2026-07-05T06:22:14.968653+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.10998v1","created_at":"2026-07-05T06:22:14.968653+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.10998","created_at":"2026-07-05T06:22:14.968653+00:00"},{"alias_kind":"pith_short_12","alias_value":"QA4BKTABBOSJ","created_at":"2026-07-05T06:22:14.968653+00:00"},{"alias_kind":"pith_short_16","alias_value":"QA4BKTABBOSJB3OB","created_at":"2026-07-05T06:22:14.968653+00:00"},{"alias_kind":"pith_short_8","alias_value":"QA4BKTAB","created_at":"2026-07-05T06:22:14.968653+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06492","citing_title":"Code2LoRA: Hypernetwork-Generated Adapters for Code Language Models under Software Evolution","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2409.04777","citing_title":"Optimization Hyper-parameter Laws for Large Language Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14635","citing_title":"SWE-QA: Can Language Models Answer Repository-level Code Questions?","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2601.00376","citing_title":"In Line with Context: Repository-Level Code Generation via Context Inlining","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19473","citing_title":"Retrieval-Augmented Generation for AI-Generated Content: A Survey","ref_index":243,"is_internal_anchor":false},{"citing_arxiv_id":"2406.00515","citing_title":"A Survey on Large Language Models for Code Generation","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11163","citing_title":"Benchmarking LLM-Based Static Analysis for Secure Smart Contract Development: Reliability, Limitations, and Potential Hybrid Solutions","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06861","citing_title":"REAgent: Requirement-Driven LLM Agents for Software Issue Resolution","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QA4BKTABBOSJB3OBZRYE2T7IFW","json":"https://pith.science/pith/QA4BKTABBOSJB3OBZRYE2T7IFW.json","graph_json":"https://pith.science/api/pith-number/QA4BKTABBOSJB3OBZRYE2T7IFW/graph.json","events_json":"https://pith.science/api/pith-number/QA4BKTABBOSJB3OBZRYE2T7IFW/events.json","paper":"https://pith.science/paper/QA4BKTAB"},"agent_actions":{"view_html":"https://pith.science/pith/QA4BKTABBOSJB3OBZRYE2T7IFW","download_json":"https://pith.science/pith/QA4BKTABBOSJB3OBZRYE2T7IFW.json","view_paper":"https://pith.science/paper/QA4BKTAB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.10998&json=true","fetch_graph":"https://pith.science/api/pith-number/QA4BKTABBOSJB3OBZRYE2T7IFW/graph.json","fetch_events":"https://pith.science/api/pith-number/QA4BKTABBOSJB3OBZRYE2T7IFW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QA4BKTABBOSJB3OBZRYE2T7IFW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QA4BKTABBOSJB3OBZRYE2T7IFW/action/storage_attestation","attest_author":"https://pith.science/pith/QA4BKTABBOSJB3OBZRYE2T7IFW/action/author_attestation","sign_citation":"https://pith.science/pith/QA4BKTABBOSJB3OBZRYE2T7IFW/action/citation_signature","submit_replication":"https://pith.science/pith/QA4BKTABBOSJB3OBZRYE2T7IFW/action/replication_record"}},"created_at":"2026-07-05T06:22:14.968653+00:00","updated_at":"2026-07-05T06:22:14.968653+00:00"}