{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IGCETUDRLO4Z6W76BNCKQELA62","short_pith_number":"pith:IGCETUDR","schema_version":"1.0","canonical_sha256":"418449d0715bb99f5bfe0b44a81160f68ec0747fe2b3c213b68fd0ec45c765d1","source":{"kind":"arxiv","id":"2402.11495","version":2},"attestation_state":"computed","paper":{"title":"Continuous Multi-Task Pre-training for Malicious URL Detection and Webpage Classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Peiyue Li, Yanbin Wang, Yifan Jia, Yiwei Liu, Yujie Li","submitted_at":"2024-02-18T07:51:20Z","abstract_excerpt":"Malicious URL detection and webpage classification are critical tasks in cybersecurity and information management. In recent years, extensive research has explored using BERT or similar language models to replace traditional machine learning methods for detecting malicious URLs and classifying webpages. While previous studies show promising results, they often apply existing language models to these tasks without accounting for the inherent differences in domain data (e.g., URLs being loosely structured and semantically sparse compared to text), leaving room for performance improvement. Furthe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.11495","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-02-18T07:51:20Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"c75a0cb6a2a35aef5ec83747db1a8b370d19a27db8aae8ecf36941e75276a695","abstract_canon_sha256":"552b1b2be77ee3b69dc7857838c65b4bae1de3f13f00f5efac8abe30dda1a29d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:56.545337Z","signature_b64":"sA/tfbH23a3e1NsogsBobzDaddxixsufRvD8gl3OZiNrPAJulYMfD8KGs3xTRLuUBWLJYct2LPsyNfRqc2QIBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"418449d0715bb99f5bfe0b44a81160f68ec0747fe2b3c213b68fd0ec45c765d1","last_reissued_at":"2026-07-05T11:08:56.544795Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:56.544795Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Continuous Multi-Task Pre-training for Malicious URL Detection and Webpage Classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Peiyue Li, Yanbin Wang, Yifan Jia, Yiwei Liu, Yujie Li","submitted_at":"2024-02-18T07:51:20Z","abstract_excerpt":"Malicious URL detection and webpage classification are critical tasks in cybersecurity and information management. In recent years, extensive research has explored using BERT or similar language models to replace traditional machine learning methods for detecting malicious URLs and classifying webpages. While previous studies show promising results, they often apply existing language models to these tasks without accounting for the inherent differences in domain data (e.g., URLs being loosely structured and semantically sparse compared to text), leaving room for performance improvement. Furthe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.11495","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.11495/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.11495","created_at":"2026-07-05T11:08:56.544871+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.11495v2","created_at":"2026-07-05T11:08:56.544871+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.11495","created_at":"2026-07-05T11:08:56.544871+00:00"},{"alias_kind":"pith_short_12","alias_value":"IGCETUDRLO4Z","created_at":"2026-07-05T11:08:56.544871+00:00"},{"alias_kind":"pith_short_16","alias_value":"IGCETUDRLO4Z6W76","created_at":"2026-07-05T11:08:56.544871+00:00"},{"alias_kind":"pith_short_8","alias_value":"IGCETUDR","created_at":"2026-07-05T11:08:56.544871+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.16449","citing_title":"From Past to Present: A Survey of Malicious URL Detection Techniques, Datasets and Code Repositories","ref_index":236,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IGCETUDRLO4Z6W76BNCKQELA62","json":"https://pith.science/pith/IGCETUDRLO4Z6W76BNCKQELA62.json","graph_json":"https://pith.science/api/pith-number/IGCETUDRLO4Z6W76BNCKQELA62/graph.json","events_json":"https://pith.science/api/pith-number/IGCETUDRLO4Z6W76BNCKQELA62/events.json","paper":"https://pith.science/paper/IGCETUDR"},"agent_actions":{"view_html":"https://pith.science/pith/IGCETUDRLO4Z6W76BNCKQELA62","download_json":"https://pith.science/pith/IGCETUDRLO4Z6W76BNCKQELA62.json","view_paper":"https://pith.science/paper/IGCETUDR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.11495&json=true","fetch_graph":"https://pith.science/api/pith-number/IGCETUDRLO4Z6W76BNCKQELA62/graph.json","fetch_events":"https://pith.science/api/pith-number/IGCETUDRLO4Z6W76BNCKQELA62/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IGCETUDRLO4Z6W76BNCKQELA62/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IGCETUDRLO4Z6W76BNCKQELA62/action/storage_attestation","attest_author":"https://pith.science/pith/IGCETUDRLO4Z6W76BNCKQELA62/action/author_attestation","sign_citation":"https://pith.science/pith/IGCETUDRLO4Z6W76BNCKQELA62/action/citation_signature","submit_replication":"https://pith.science/pith/IGCETUDRLO4Z6W76BNCKQELA62/action/replication_record"}},"created_at":"2026-07-05T11:08:56.544871+00:00","updated_at":"2026-07-05T11:08:56.544871+00:00"}