{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BHIGNLEXTYLONJQIEBFS46A65M","short_pith_number":"pith:BHIGNLEX","schema_version":"1.0","canonical_sha256":"09d066ac979e16e6a608204b2e781eeb23deaef81fe4e7d07c8bdac106f9d1bc","source":{"kind":"arxiv","id":"2402.14261","version":1},"attestation_state":"computed","paper":{"title":"Copilot Evaluation Harness: Evaluating LLM-Guided Software Programming","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Aaron Chan, Anisha Agarwal, Jinu Jang, Michele Tufano, Neel Sundaresan, Roshanak Zilouchian Moghaddam, Shaun Miller, Shubham Chandel, Yevhen Mohylevskyy","submitted_at":"2024-02-22T03:51:34Z","abstract_excerpt":"The integration of Large Language Models (LLMs) into Development Environments (IDEs) has become a focal point in modern software development. LLMs such as OpenAI GPT-3.5/4 and Code Llama offer the potential to significantly augment developer productivity by serving as intelligent, chat-driven programming assistants. However, utilizing LLMs out of the box is unlikely to be optimal for any given scenario. Rather, each system requires the LLM to be honed to its set of heuristics to ensure the best performance. In this paper, we introduce the Copilot evaluation harness: a set of data and tools for"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14261","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-02-22T03:51:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a2e4f80b5d773e206e7d1ea7c6b8df73711a9d76cdfc35c009d8f81355106d5b","abstract_canon_sha256":"2dc6ad6c198a24e3f633fd814cea1a29e0317892327c6e6b41a81109471c6b0f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:48:07.814763Z","signature_b64":"QO5Ek8IeZoOwJ7hstc85zAIwdxTJUypM8iLX8ooq2S8CpD3P79fAoqrg1NNBdHOHay4L2+w1w0fFUmZAdMoVCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"09d066ac979e16e6a608204b2e781eeb23deaef81fe4e7d07c8bdac106f9d1bc","last_reissued_at":"2026-07-05T07:48:07.814283Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:48:07.814283Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Copilot Evaluation Harness: Evaluating LLM-Guided Software Programming","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Aaron Chan, Anisha Agarwal, Jinu Jang, Michele Tufano, Neel Sundaresan, Roshanak Zilouchian Moghaddam, Shaun Miller, Shubham Chandel, Yevhen Mohylevskyy","submitted_at":"2024-02-22T03:51:34Z","abstract_excerpt":"The integration of Large Language Models (LLMs) into Development Environments (IDEs) has become a focal point in modern software development. LLMs such as OpenAI GPT-3.5/4 and Code Llama offer the potential to significantly augment developer productivity by serving as intelligent, chat-driven programming assistants. However, utilizing LLMs out of the box is unlikely to be optimal for any given scenario. Rather, each system requires the LLM to be honed to its set of heuristics to ensure the best performance. In this paper, we introduce the Copilot evaluation harness: a set of data and tools for"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14261","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14261/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14261","created_at":"2026-07-05T07:48:07.814351+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14261v1","created_at":"2026-07-05T07:48:07.814351+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14261","created_at":"2026-07-05T07:48:07.814351+00:00"},{"alias_kind":"pith_short_12","alias_value":"BHIGNLEXTYLO","created_at":"2026-07-05T07:48:07.814351+00:00"},{"alias_kind":"pith_short_16","alias_value":"BHIGNLEXTYLONJQI","created_at":"2026-07-05T07:48:07.814351+00:00"},{"alias_kind":"pith_short_8","alias_value":"BHIGNLEX","created_at":"2026-07-05T07:48:07.814351+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03137","citing_title":"Think-Before-Speak: From Internal Evaluation to Public Expression in Multi-Agent Social Simulation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21623","citing_title":"OjaKV: Context-Aware Online Low-Rank KV Cache Compression","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BHIGNLEXTYLONJQIEBFS46A65M","json":"https://pith.science/pith/BHIGNLEXTYLONJQIEBFS46A65M.json","graph_json":"https://pith.science/api/pith-number/BHIGNLEXTYLONJQIEBFS46A65M/graph.json","events_json":"https://pith.science/api/pith-number/BHIGNLEXTYLONJQIEBFS46A65M/events.json","paper":"https://pith.science/paper/BHIGNLEX"},"agent_actions":{"view_html":"https://pith.science/pith/BHIGNLEXTYLONJQIEBFS46A65M","download_json":"https://pith.science/pith/BHIGNLEXTYLONJQIEBFS46A65M.json","view_paper":"https://pith.science/paper/BHIGNLEX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14261&json=true","fetch_graph":"https://pith.science/api/pith-number/BHIGNLEXTYLONJQIEBFS46A65M/graph.json","fetch_events":"https://pith.science/api/pith-number/BHIGNLEXTYLONJQIEBFS46A65M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BHIGNLEXTYLONJQIEBFS46A65M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BHIGNLEXTYLONJQIEBFS46A65M/action/storage_attestation","attest_author":"https://pith.science/pith/BHIGNLEXTYLONJQIEBFS46A65M/action/author_attestation","sign_citation":"https://pith.science/pith/BHIGNLEXTYLONJQIEBFS46A65M/action/citation_signature","submit_replication":"https://pith.science/pith/BHIGNLEXTYLONJQIEBFS46A65M/action/replication_record"}},"created_at":"2026-07-05T07:48:07.814351+00:00","updated_at":"2026-07-05T07:48:07.814351+00:00"}