{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:4IPOKMJKMNFBR6KT3HSUEQOANU","short_pith_number":"pith:4IPOKMJK","schema_version":"1.0","canonical_sha256":"e21ee5312a634a18f953d9e54241c06d2ac2a8a438558fe4bbe6d0847e2c3071","source":{"kind":"arxiv","id":"1909.09436","version":3},"attestation_state":"computed","paper":{"title":"CodeSearchNet Challenge: Evaluating the State of Semantic Code Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Releasing the CodeSearchNet Corpus of 6 million functions and a challenge with 99 annotated queries enables evaluation of semantic code search across six languages.","cross_cats":["cs.IR","cs.SE","stat.ML"],"primary_cat":"cs.LG","authors_text":"Hamel Husain, Ho-Hsiang Wu, Marc Brockschmidt, Miltiadis Allamanis, Tiferet Gazit","submitted_at":"2019-09-20T11:52:45Z","abstract_excerpt":"Semantic code search is the task of retrieving relevant code given a natural language query. While related to other information retrieval tasks, it requires bridging the gap between the language used in code (often abbreviated and highly technical) and natural language more suitable to describe vague concepts and ideas.\n  To enable evaluation of progress on code search, we are releasing the CodeSearchNet Corpus and are presenting the CodeSearchNet Challenge, which consists of 99 natural language queries with about 4k expert relevance annotations of likely results from CodeSearchNet Corpus. The"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":true,"formal_links_present":true},"canonical_record":{"source":{"id":"1909.09436","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-09-20T11:52:45Z","cross_cats_sorted":["cs.IR","cs.SE","stat.ML"],"title_canon_sha256":"30eb13f7e19b295d94df00528fcc5f3e2875ebea204bb9870398d335e448954e","abstract_canon_sha256":"33e2d50bac7613c65aff732583ff944c4a2428b24e64c4b1b2072a205f86e5cc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:08:24.850341Z","signature_b64":"P5z0iH2qvPjTLeTi0rvejwny/0E8w+muQVAgPaDlO31pyOuhgki9c4tFI42pzJ42e6G5uTuavgh3lyr8zILUAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e21ee5312a634a18f953d9e54241c06d2ac2a8a438558fe4bbe6d0847e2c3071","last_reissued_at":"2026-07-05T01:08:24.849922Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:08:24.849922Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CodeSearchNet Challenge: Evaluating the State of Semantic Code Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Releasing the CodeSearchNet Corpus of 6 million functions and a challenge with 99 annotated queries enables evaluation of semantic code search across six languages.","cross_cats":["cs.IR","cs.SE","stat.ML"],"primary_cat":"cs.LG","authors_text":"Hamel Husain, Ho-Hsiang Wu, Marc Brockschmidt, Miltiadis Allamanis, Tiferet Gazit","submitted_at":"2019-09-20T11:52:45Z","abstract_excerpt":"Semantic code search is the task of retrieving relevant code given a natural language query. While related to other information retrieval tasks, it requires bridging the gap between the language used in code (often abbreviated and highly technical) and natural language more suitable to describe vague concepts and ideas.\n  To enable evaluation of progress on code search, we are releasing the CodeSearchNet Corpus and are presenting the CodeSearchNet Challenge, which consists of 99 natural language queries with about 4k expert relevance annotations of likely results from CodeSearchNet Corpus. The"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"To enable evaluation of progress on code search, we are releasing the CodeSearchNet Corpus and are presenting the CodeSearchNet Challenge, which consists of 99 natural language queries with about 4k expert relevance annotations of likely results from CodeSearchNet Corpus.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The assumption that mechanically scraped and preprocessed function documentation yields sufficiently accurate and representative natural-language queries, and that the expert annotations are consistent and unbiased measures of relevance.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Releases a large multi-language code corpus and expert-annotated challenge to benchmark semantic code search.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Releasing the CodeSearchNet Corpus of 6 million functions and a challenge with 99 annotated queries enables evaluation of semantic code search across six languages.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"8dc5cff80129a1bba3a4933a945206601984ac61bb24dea5c405d5ff3fe72ecc"},"source":{"id":"1909.09436","kind":"arxiv","version":3},"verdict":{"id":"cf2c739e-89a0-47ee-99eb-ff88f7c7f9ea","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-12T16:00:39.026015Z","strongest_claim":"To enable evaluation of progress on code search, we are releasing the CodeSearchNet Corpus and are presenting the CodeSearchNet Challenge, which consists of 99 natural language queries with about 4k expert relevance annotations of likely results from CodeSearchNet Corpus.","one_line_summary":"Releases a large multi-language code corpus and expert-annotated challenge to benchmark semantic code search.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The assumption that mechanically scraped and preprocessed function documentation yields sufficiently accurate and representative natural-language queries, and that the expert annotations are consistent and unbiased measures of relevance.","pith_extraction_headline":"Releasing the CodeSearchNet Corpus of 6 million functions and a challenge with 99 annotated queries enables evaluation of semantic code search across six languages."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.09436/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":26,"sample":[{"doi":"","year":2018,"title":"Miltiadis Allamanis. 2018. The Adverse Effects of Code Duplication in Machine Learning Models of Code. arXiv preprint arXiv:1812.06469 (2018)","work_id":"ec049e9a-4ab6-434d-8752-48618636162a","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2018,"title":"Miltiadis Allamanis, Earl T Barr, Premkumar Devanbu, and Charles Sutton. 2018. A survey of machine learning for big code and naturalness. ACM Computing Surveys (CSUR) 51, 4 (2018), 81","work_id":"7f0977f1-2213-4528-a924-89e2f327a516","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2016,"title":"Miltiadis Allamanis, Hao Peng, and Charles Sutton. 2016. A Convolutional Attention Network for Extreme Summarization of Source Code. In Proceedings of the International Conference on Machine Learning ","work_id":"7aeef716-5af6-4a34-b2a0-fe454133cb79","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2018,"title":"code2seq: Generating Sequences from Structured Representations of Code","work_id":"f4d5c460-f108-4fb4-97a4-6109f749e7ba","ref_index":4,"cited_arxiv_id":"1808.01400","is_internal_anchor":false},{"doi":"","year":2017,"title":"A parallel corpus of Python functions and documentation strings for automated code documentation and code generation","work_id":"3484b4d3-268c-4c6a-8eb5-315f36c3de9a","ref_index":5,"cited_arxiv_id":"1707.02275","is_internal_anchor":false}],"resolved_work":26,"snapshot_sha256":"daa9c241e1b4db220b931a723df354d3368eb96327533d25986ed8fda92c5753","internal_anchors":1},"formal_canon":{"evidence_count":2,"snapshot_sha256":"d95495fcd98942c9d51b78e2f699401dad5ae1529d67a2d5d341e0a79c14ee53"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.09436","created_at":"2026-07-05T01:08:24.849980+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.09436v3","created_at":"2026-07-05T01:08:24.849980+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.09436","created_at":"2026-07-05T01:08:24.849980+00:00"},{"alias_kind":"pith_short_12","alias_value":"4IPOKMJKMNFB","created_at":"2026-07-05T01:08:24.849980+00:00"},{"alias_kind":"pith_short_16","alias_value":"4IPOKMJKMNFBR6KT","created_at":"2026-07-05T01:08:24.849980+00:00"},{"alias_kind":"pith_short_8","alias_value":"4IPOKMJK","created_at":"2026-07-05T01:08:24.849980+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":81,"internal_anchor_count":81,"sample":[{"citing_arxiv_id":"2607.08009","citing_title":"From Execution to Education: A Bloom-Aligned Framework for Measuring Educational Control in LLMs","ref_index":75,"is_internal_anchor":true},{"citing_arxiv_id":"2606.07519","citing_title":"Bidirectional Small-Granularity Search between Code and Text","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23877","citing_title":"JupOtter: Cell-Level Bug Detection in Jupyter Notebooks","ref_index":26,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21647","citing_title":"ConcernBERT: Learning Responsibilities Using Class Membership","ref_index":67,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19988","citing_title":"Repository-Level Solidity Code Generation with Large Language Models: From Prompting to Fine-Tuning","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2605.16046","citing_title":"XSearch: Explainable Code Search via Concept-to-Code Alignment","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19100","citing_title":"AMALIA-VL: A Native European Portuguese Open-Source Vision and Language Model","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01425","citing_title":"Agent4cs: A Multi-agent System for Code Summarization in Large Hierarchical Codebases","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11864","citing_title":"CORE-Bench: A Comprehensive Benchmark for Code Retrieval in the Era of Agentic Coding","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12620","citing_title":"HybridCodeAuthorship: A Benchmark Dataset for Line-Level Code Authorship Detection","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2606.10846","citing_title":"Securing Code Understanding: Detecting Natural Backdoor Vulnerability in Code Language Models","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19100","citing_title":"AMALIA-VL: A Native European Portuguese Open-Source Vision and Language Model","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2606.08151","citing_title":"Decision-Aware Memory Cards: Counterfactual-Inspired Context Selection and Compression for Tool-Using LLM Agents","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2606.07297","citing_title":"SWE-Explore: Benchmarking How Coding Agents Explore Repositories","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06735","citing_title":"A Geometric Account of Activation Steering through Angle-Norm Decomposition","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04273","citing_title":"Characterizing initial human-AI proof formalization workflows","ref_index":134,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03990","citing_title":"Neuron Populations Exhibit Divergent Selectivity with Scale","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2605.31520","citing_title":"Separating Secrets from Placeholders: A Hybrid CNN-CodeBERT Framework for Three-Class Credential Leakage Detection","ref_index":29,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27747","citing_title":"UNICS: Multilingual Code Search via Unified Pseudocode and Contrastive Transfer Learning","ref_index":26,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31272","citing_title":"The Decomposition Is the Fingerprint: Per-Component Identity for Agent Skills","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2604.27676","citing_title":"Users' Activity Logs: the Good, the Bad, the Misconception, and the Disastrous","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19100","citing_title":"AMALIA-VL: A Native European Portuguese Open-Source Vision and Language Model","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2605.26144","citing_title":"VISTA: An End-to-End Benchmark for Visual Spec-to-Web-App Coding Agents","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2606.28367","citing_title":"Beyond the Reranker: Do RAG Retrieval Enhancements Help Once a Strong Reranker Is Present?","ref_index":47,"is_internal_anchor":true},{"citing_arxiv_id":"2606.29538","citing_title":"RESOURCE2SKILL: Distilling Executable Agent Skills from Human-Created Multimodal Resources","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":2,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4IPOKMJKMNFBR6KT3HSUEQOANU","json":"https://pith.science/pith/4IPOKMJKMNFBR6KT3HSUEQOANU.json","graph_json":"https://pith.science/api/pith-number/4IPOKMJKMNFBR6KT3HSUEQOANU/graph.json","events_json":"https://pith.science/api/pith-number/4IPOKMJKMNFBR6KT3HSUEQOANU/events.json","paper":"https://pith.science/paper/4IPOKMJK"},"agent_actions":{"view_html":"https://pith.science/pith/4IPOKMJKMNFBR6KT3HSUEQOANU","download_json":"https://pith.science/pith/4IPOKMJKMNFBR6KT3HSUEQOANU.json","view_paper":"https://pith.science/paper/4IPOKMJK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.09436&json=true","fetch_graph":"https://pith.science/api/pith-number/4IPOKMJKMNFBR6KT3HSUEQOANU/graph.json","fetch_events":"https://pith.science/api/pith-number/4IPOKMJKMNFBR6KT3HSUEQOANU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4IPOKMJKMNFBR6KT3HSUEQOANU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4IPOKMJKMNFBR6KT3HSUEQOANU/action/storage_attestation","attest_author":"https://pith.science/pith/4IPOKMJKMNFBR6KT3HSUEQOANU/action/author_attestation","sign_citation":"https://pith.science/pith/4IPOKMJKMNFBR6KT3HSUEQOANU/action/citation_signature","submit_replication":"https://pith.science/pith/4IPOKMJKMNFBR6KT3HSUEQOANU/action/replication_record"}},"created_at":"2026-07-05T01:08:24.849980+00:00","updated_at":"2026-07-05T01:08:24.849980+00:00"}