{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UMSSTQM3EASAFJYTEQ7C6LKGV6","short_pith_number":"pith:UMSSTQM3","schema_version":"1.0","canonical_sha256":"a32529c19b202402a713243e2f2d46af8268d34cf4151ecd1e8d2a89feb9f281","source":{"kind":"arxiv","id":"2503.20803","version":2},"attestation_state":"computed","paper":{"title":"Leveraging VAE-Derived Latent Spaces for Enhanced Malware Detection with Machine Learning Classifiers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Bamidele Ajayi, Basel Barakat, Ken McGarry","submitted_at":"2025-03-24T14:44:55Z","abstract_excerpt":"This paper assesses the performance of five machine learning classifiers: Decision Tree, Naive Bayes, LightGBM, Logistic Regression, and Random Forest using latent representations learned by a Variational Autoencoder from malware datasets. Results from the experiments conducted on different training-test splits with different random seeds reveal that all the models perform well in detecting malware with ensemble methods (LightGBM and Random Forest) performing slightly better than the rest. In addition, the use of latent features reduces the computational cost of the model and the need for exte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.20803","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2025-03-24T14:44:55Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"8130827b3425cd3f1bd55f56037961a663085acfd6c631eba805255a9238b95e","abstract_canon_sha256":"dc2d910f6ec0ec5a8cf5c2168a931757c101f4cadc463a745c5a486a6bea2578"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:56:16.634121Z","signature_b64":"Wln6uWeldCGv/MxaJliZXWZCDpYYY4ihJ1CgXFH3LxESA+yoRuhX1zvQDwxtHrNJ7bJrrGZy7xoRF5SkqwOSBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a32529c19b202402a713243e2f2d46af8268d34cf4151ecd1e8d2a89feb9f281","last_reissued_at":"2026-07-05T10:56:16.633605Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:56:16.633605Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Leveraging VAE-Derived Latent Spaces for Enhanced Malware Detection with Machine Learning Classifiers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Bamidele Ajayi, Basel Barakat, Ken McGarry","submitted_at":"2025-03-24T14:44:55Z","abstract_excerpt":"This paper assesses the performance of five machine learning classifiers: Decision Tree, Naive Bayes, LightGBM, Logistic Regression, and Random Forest using latent representations learned by a Variational Autoencoder from malware datasets. Results from the experiments conducted on different training-test splits with different random seeds reveal that all the models perform well in detecting malware with ensemble methods (LightGBM and Random Forest) performing slightly better than the rest. In addition, the use of latent features reduces the computational cost of the model and the need for exte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.20803","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.20803/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.20803","created_at":"2026-07-05T10:56:16.633664+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.20803v2","created_at":"2026-07-05T10:56:16.633664+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.20803","created_at":"2026-07-05T10:56:16.633664+00:00"},{"alias_kind":"pith_short_12","alias_value":"UMSSTQM3EASA","created_at":"2026-07-05T10:56:16.633664+00:00"},{"alias_kind":"pith_short_16","alias_value":"UMSSTQM3EASAFJYT","created_at":"2026-07-05T10:56:16.633664+00:00"},{"alias_kind":"pith_short_8","alias_value":"UMSSTQM3","created_at":"2026-07-05T10:56:16.633664+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.16952","citing_title":"Evaluating Ensemble and Deep Learning Models for Static Malware Detection with Dimensionality Reduction Using the EMBER Dataset","ref_index":2025,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UMSSTQM3EASAFJYTEQ7C6LKGV6","json":"https://pith.science/pith/UMSSTQM3EASAFJYTEQ7C6LKGV6.json","graph_json":"https://pith.science/api/pith-number/UMSSTQM3EASAFJYTEQ7C6LKGV6/graph.json","events_json":"https://pith.science/api/pith-number/UMSSTQM3EASAFJYTEQ7C6LKGV6/events.json","paper":"https://pith.science/paper/UMSSTQM3"},"agent_actions":{"view_html":"https://pith.science/pith/UMSSTQM3EASAFJYTEQ7C6LKGV6","download_json":"https://pith.science/pith/UMSSTQM3EASAFJYTEQ7C6LKGV6.json","view_paper":"https://pith.science/paper/UMSSTQM3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.20803&json=true","fetch_graph":"https://pith.science/api/pith-number/UMSSTQM3EASAFJYTEQ7C6LKGV6/graph.json","fetch_events":"https://pith.science/api/pith-number/UMSSTQM3EASAFJYTEQ7C6LKGV6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UMSSTQM3EASAFJYTEQ7C6LKGV6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UMSSTQM3EASAFJYTEQ7C6LKGV6/action/storage_attestation","attest_author":"https://pith.science/pith/UMSSTQM3EASAFJYTEQ7C6LKGV6/action/author_attestation","sign_citation":"https://pith.science/pith/UMSSTQM3EASAFJYTEQ7C6LKGV6/action/citation_signature","submit_replication":"https://pith.science/pith/UMSSTQM3EASAFJYTEQ7C6LKGV6/action/replication_record"}},"created_at":"2026-07-05T10:56:16.633664+00:00","updated_at":"2026-07-05T10:56:16.633664+00:00"}