{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:4E2KD5DVJ2TJAQIW7BHUXHP7FU","short_pith_number":"pith:4E2KD5DV","schema_version":"1.0","canonical_sha256":"e134a1f4754ea6904116f84f4b9dff2d1f0fc0252986d07a1be91bdb95790dbe","source":{"kind":"arxiv","id":"2602.16304","version":3},"attestation_state":"computed","paper":{"title":"An Evaluation of Large Language Models for Detection of Malicious Python Packages","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CR","authors_text":"Abdullah Al Jahid, Ahmed Ryan, Akond Ashfaque Ur Rahman, Ibrahim Khalil, Md Erfan, Md Rayhanur Rahman, Sungbin Park","submitted_at":"2026-02-18T09:36:46Z","abstract_excerpt":"Modern software development relies on open-source package repositories. Attackers use these to distribute malicious packages. Large Language Models (LLMs) can automatically detect these packages, but their ability to pinpoint specific malicious behaviors remains unclear. We evaluate 13 LLMs on two tasks using a dataset of 4,070 PyPI packages (370 malicious, 3,700 benign). The first task detects whether a package is malicious. The second identifies specific malicious indicators (lines of code). We evaluate each LLM across five prompt strategies and three temperatures.\n  For the first task, LLMs"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.16304","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2026-02-18T09:36:46Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"fcc40e11db0dd4a4c43e8681aad5081d3628ffe5c872483038b72a2999ba0d2c","abstract_canon_sha256":"0a50dbe1a480bde87d0ecbbfea946978b7f4357baad56f7563e7c6eb977242b6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-15T00:21:16.576611Z","signature_b64":"36oygDlqhOsvA5T0lToAuhuH8/ORbRyuy4x8DIokvwkz7pVi0pD6PaUvwAOd1SmDW7QJyN597bVjXgJJtB4TBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e134a1f4754ea6904116f84f4b9dff2d1f0fc0252986d07a1be91bdb95790dbe","last_reissued_at":"2026-07-15T00:21:16.575709Z","signature_status":"signed_v1","first_computed_at":"2026-07-15T00:21:16.575709Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Evaluation of Large Language Models for Detection of Malicious Python Packages","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CR","authors_text":"Abdullah Al Jahid, Ahmed Ryan, Akond Ashfaque Ur Rahman, Ibrahim Khalil, Md Erfan, Md Rayhanur Rahman, Sungbin Park","submitted_at":"2026-02-18T09:36:46Z","abstract_excerpt":"Modern software development relies on open-source package repositories. Attackers use these to distribute malicious packages. Large Language Models (LLMs) can automatically detect these packages, but their ability to pinpoint specific malicious behaviors remains unclear. We evaluate 13 LLMs on two tasks using a dataset of 4,070 PyPI packages (370 malicious, 3,700 benign). The first task detects whether a package is malicious. The second identifies specific malicious indicators (lines of code). We evaluate each LLM across five prompt strategies and three temperatures.\n  For the first task, LLMs"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.16304","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.16304/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.16304","created_at":"2026-07-15T00:21:16.576125+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.16304v3","created_at":"2026-07-15T00:21:16.576125+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.16304","created_at":"2026-07-15T00:21:16.576125+00:00"},{"alias_kind":"pith_short_12","alias_value":"4E2KD5DVJ2TJ","created_at":"2026-07-15T00:21:16.576125+00:00"},{"alias_kind":"pith_short_16","alias_value":"4E2KD5DVJ2TJAQIW","created_at":"2026-07-15T00:21:16.576125+00:00"},{"alias_kind":"pith_short_8","alias_value":"4E2KD5DV","created_at":"2026-07-15T00:21:16.576125+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.22601","citing_title":"From Natural Language to Verified Code: Toward AI Assisted Problem-to-Code Generation with Dafny-Based Formal Verification","ref_index":72,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4E2KD5DVJ2TJAQIW7BHUXHP7FU","json":"https://pith.science/pith/4E2KD5DVJ2TJAQIW7BHUXHP7FU.json","graph_json":"https://pith.science/api/pith-number/4E2KD5DVJ2TJAQIW7BHUXHP7FU/graph.json","events_json":"https://pith.science/api/pith-number/4E2KD5DVJ2TJAQIW7BHUXHP7FU/events.json","paper":"https://pith.science/paper/4E2KD5DV"},"agent_actions":{"view_html":"https://pith.science/pith/4E2KD5DVJ2TJAQIW7BHUXHP7FU","download_json":"https://pith.science/pith/4E2KD5DVJ2TJAQIW7BHUXHP7FU.json","view_paper":"https://pith.science/paper/4E2KD5DV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.16304&json=true","fetch_graph":"https://pith.science/api/pith-number/4E2KD5DVJ2TJAQIW7BHUXHP7FU/graph.json","fetch_events":"https://pith.science/api/pith-number/4E2KD5DVJ2TJAQIW7BHUXHP7FU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4E2KD5DVJ2TJAQIW7BHUXHP7FU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4E2KD5DVJ2TJAQIW7BHUXHP7FU/action/storage_attestation","attest_author":"https://pith.science/pith/4E2KD5DVJ2TJAQIW7BHUXHP7FU/action/author_attestation","sign_citation":"https://pith.science/pith/4E2KD5DVJ2TJAQIW7BHUXHP7FU/action/citation_signature","submit_replication":"https://pith.science/pith/4E2KD5DVJ2TJAQIW7BHUXHP7FU/action/replication_record"}},"created_at":"2026-07-15T00:21:16.576125+00:00","updated_at":"2026-07-15T00:21:16.576125+00:00"}