{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2U5NS2QOGWSD5QAODMM3EOUX4A","short_pith_number":"pith:2U5NS2QO","schema_version":"1.0","canonical_sha256":"d53ad96a0e35a43ec00e1b19b23a97e021bf0a4c3dc155500637e6ebf9cb4f2a","source":{"kind":"arxiv","id":"2409.15228","version":3},"attestation_state":"computed","paper":{"title":"A Comprehensive Framework for Evaluating API-oriented Code Generation in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Pengfei He, Shaowei Wang, Tse-Hsun Chen, Yixi Wu, Yuan Tian, Zehao Wang","submitted_at":"2024-09-23T17:22:09Z","abstract_excerpt":"Large language models (LLMs) like GitHub Copilot and ChatGPT have emerged as powerful tools for code generation, significantly enhancing productivity and accelerating software development. However, existing benchmarks primarily focus on general code generation without considering API-oriented code generation, i.e., generating code that invokes APIs from specific libraries. Given the growing demand for API-oriented code generation, there is a pressing need for a systematic and automated approach to evaluate LLM on API-oriented code generation. To address this gap, we propose AutoAPIEval, a ligh"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.15228","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-09-23T17:22:09Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d643de1b4401afee1ecb8097bd463c0e8bd4a6dca458bbad97561378803447f1","abstract_canon_sha256":"2c3d1fb9f3b2252f0cd60c4fa82f55d918d978e1199ac690100dd0031ee15341"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:09.948747Z","signature_b64":"OVIh4Qg+hRGnRrGqM0G/Hrsc640qZlWF22onv/RTG+AicR12XoIAZTpwfkMThc+h5RD9U6Zid+8LG85HvRNlCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d53ad96a0e35a43ec00e1b19b23a97e021bf0a4c3dc155500637e6ebf9cb4f2a","last_reissued_at":"2026-07-05T09:12:09.948223Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:09.948223Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comprehensive Framework for Evaluating API-oriented Code Generation in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Pengfei He, Shaowei Wang, Tse-Hsun Chen, Yixi Wu, Yuan Tian, Zehao Wang","submitted_at":"2024-09-23T17:22:09Z","abstract_excerpt":"Large language models (LLMs) like GitHub Copilot and ChatGPT have emerged as powerful tools for code generation, significantly enhancing productivity and accelerating software development. However, existing benchmarks primarily focus on general code generation without considering API-oriented code generation, i.e., generating code that invokes APIs from specific libraries. Given the growing demand for API-oriented code generation, there is a pressing need for a systematic and automated approach to evaluate LLM on API-oriented code generation. To address this gap, we propose AutoAPIEval, a ligh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.15228","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.15228/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.15228","created_at":"2026-07-05T09:12:09.948273+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.15228v3","created_at":"2026-07-05T09:12:09.948273+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.15228","created_at":"2026-07-05T09:12:09.948273+00:00"},{"alias_kind":"pith_short_12","alias_value":"2U5NS2QOGWSD","created_at":"2026-07-05T09:12:09.948273+00:00"},{"alias_kind":"pith_short_16","alias_value":"2U5NS2QOGWSD5QAO","created_at":"2026-07-05T09:12:09.948273+00:00"},{"alias_kind":"pith_short_8","alias_value":"2U5NS2QO","created_at":"2026-07-05T09:12:09.948273+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31135","citing_title":"R+R: Reassessing Java Security API Misuse in Current LLMs: A Replication on JCA and JSSE APIs with External Security Knowledge","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03622","citing_title":"Toward Executable Repository-Level Code Generation via Environment Alignment","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05267","citing_title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09515","citing_title":"When LLMs Lag Behind: Knowledge Conflicts from Evolving APIs in Code Generation","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2U5NS2QOGWSD5QAODMM3EOUX4A","json":"https://pith.science/pith/2U5NS2QOGWSD5QAODMM3EOUX4A.json","graph_json":"https://pith.science/api/pith-number/2U5NS2QOGWSD5QAODMM3EOUX4A/graph.json","events_json":"https://pith.science/api/pith-number/2U5NS2QOGWSD5QAODMM3EOUX4A/events.json","paper":"https://pith.science/paper/2U5NS2QO"},"agent_actions":{"view_html":"https://pith.science/pith/2U5NS2QOGWSD5QAODMM3EOUX4A","download_json":"https://pith.science/pith/2U5NS2QOGWSD5QAODMM3EOUX4A.json","view_paper":"https://pith.science/paper/2U5NS2QO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.15228&json=true","fetch_graph":"https://pith.science/api/pith-number/2U5NS2QOGWSD5QAODMM3EOUX4A/graph.json","fetch_events":"https://pith.science/api/pith-number/2U5NS2QOGWSD5QAODMM3EOUX4A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2U5NS2QOGWSD5QAODMM3EOUX4A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2U5NS2QOGWSD5QAODMM3EOUX4A/action/storage_attestation","attest_author":"https://pith.science/pith/2U5NS2QOGWSD5QAODMM3EOUX4A/action/author_attestation","sign_citation":"https://pith.science/pith/2U5NS2QOGWSD5QAODMM3EOUX4A/action/citation_signature","submit_replication":"https://pith.science/pith/2U5NS2QOGWSD5QAODMM3EOUX4A/action/replication_record"}},"created_at":"2026-07-05T09:12:09.948273+00:00","updated_at":"2026-07-05T09:12:09.948273+00:00"}