{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VUUK4FRJZIT5UQIQOKIR34TVHL","short_pith_number":"pith:VUUK4FRJ","schema_version":"1.0","canonical_sha256":"ad28ae1629ca27da411072911df2753acef7c7c5e2f44512d234ade5ccbdbd46","source":{"kind":"arxiv","id":"2408.03354","version":3},"attestation_state":"computed","paper":{"title":"The Use of Large Language Models (LLM) for Cyber Threat Intelligence (CTI) in Cybercrime Forums","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Eric Clay, Estelle Ruellan, Isa-May Beauchamp, Masarah Paquet-Clouston, Serge-Olivier Paquette, Vanessa Clairoux-Trepanier","submitted_at":"2024-08-06T09:15:25Z","abstract_excerpt":"Large language models (LLMs) can be used to analyze cyber threat intelligence (CTI) data from cybercrime forums, which contain extensive information and key discussions about emerging cyber threats. However, to date, the level of accuracy and efficiency of LLMs for such critical tasks has yet to be thoroughly evaluated. Hence, this study assesses the performance of an LLM system built on the OpenAI GPT-3.5-turbo model [8] to extract CTI information. To do so, a random sample of more than 700 daily conversations from three cybercrime forums - XSS, Exploit_in, and RAMP - was extracted, and the L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.03354","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-08-06T09:15:25Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"5f252bb3a5a46af4fa5f8b706240ba321fd687e6c5888c7f4215d7dd255d4508","abstract_canon_sha256":"1d48f03fe5eb7d28f82dcfd297b25258a24f03038a2a80dd260e93061bf32443"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:55.586359Z","signature_b64":"lHiN+gxHtDsGyrjCyi9cVIsF2/KTvT6ddU3P3oHyl3Cux5/+PTwIziIVaIwG3gobwkSNURtODsgElhibIcKjDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ad28ae1629ca27da411072911df2753acef7c7c5e2f44512d234ade5ccbdbd46","last_reissued_at":"2026-07-05T09:13:55.585837Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:55.585837Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Use of Large Language Models (LLM) for Cyber Threat Intelligence (CTI) in Cybercrime Forums","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Eric Clay, Estelle Ruellan, Isa-May Beauchamp, Masarah Paquet-Clouston, Serge-Olivier Paquette, Vanessa Clairoux-Trepanier","submitted_at":"2024-08-06T09:15:25Z","abstract_excerpt":"Large language models (LLMs) can be used to analyze cyber threat intelligence (CTI) data from cybercrime forums, which contain extensive information and key discussions about emerging cyber threats. However, to date, the level of accuracy and efficiency of LLMs for such critical tasks has yet to be thoroughly evaluated. Hence, this study assesses the performance of an LLM system built on the OpenAI GPT-3.5-turbo model [8] to extract CTI information. To do so, a random sample of more than 700 daily conversations from three cybercrime forums - XSS, Exploit_in, and RAMP - was extracted, and the L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.03354","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.03354/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.03354","created_at":"2026-07-05T09:13:55.585900+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.03354v3","created_at":"2026-07-05T09:13:55.585900+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.03354","created_at":"2026-07-05T09:13:55.585900+00:00"},{"alias_kind":"pith_short_12","alias_value":"VUUK4FRJZIT5","created_at":"2026-07-05T09:13:55.585900+00:00"},{"alias_kind":"pith_short_16","alias_value":"VUUK4FRJZIT5UQIQ","created_at":"2026-07-05T09:13:55.585900+00:00"},{"alias_kind":"pith_short_8","alias_value":"VUUK4FRJ","created_at":"2026-07-05T09:13:55.585900+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.06921","citing_title":"Neuro-Symbolic AI for Cybersecurity: State of the Art, Challenges, and Opportunities","ref_index":91,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VUUK4FRJZIT5UQIQOKIR34TVHL","json":"https://pith.science/pith/VUUK4FRJZIT5UQIQOKIR34TVHL.json","graph_json":"https://pith.science/api/pith-number/VUUK4FRJZIT5UQIQOKIR34TVHL/graph.json","events_json":"https://pith.science/api/pith-number/VUUK4FRJZIT5UQIQOKIR34TVHL/events.json","paper":"https://pith.science/paper/VUUK4FRJ"},"agent_actions":{"view_html":"https://pith.science/pith/VUUK4FRJZIT5UQIQOKIR34TVHL","download_json":"https://pith.science/pith/VUUK4FRJZIT5UQIQOKIR34TVHL.json","view_paper":"https://pith.science/paper/VUUK4FRJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.03354&json=true","fetch_graph":"https://pith.science/api/pith-number/VUUK4FRJZIT5UQIQOKIR34TVHL/graph.json","fetch_events":"https://pith.science/api/pith-number/VUUK4FRJZIT5UQIQOKIR34TVHL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VUUK4FRJZIT5UQIQOKIR34TVHL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VUUK4FRJZIT5UQIQOKIR34TVHL/action/storage_attestation","attest_author":"https://pith.science/pith/VUUK4FRJZIT5UQIQOKIR34TVHL/action/author_attestation","sign_citation":"https://pith.science/pith/VUUK4FRJZIT5UQIQOKIR34TVHL/action/citation_signature","submit_replication":"https://pith.science/pith/VUUK4FRJZIT5UQIQOKIR34TVHL/action/replication_record"}},"created_at":"2026-07-05T09:13:55.585900+00:00","updated_at":"2026-07-05T09:13:55.585900+00:00"}