{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BNZQWC3JMMSDHQY36P7DTUVBTN","short_pith_number":"pith:BNZQWC3J","schema_version":"1.0","canonical_sha256":"0b730b0b69632433c31bf3fe39d2a19b55659e67448ce031b3659faf500cf27b","source":{"kind":"arxiv","id":"2409.06072","version":1},"attestation_state":"computed","paper":{"title":"DetoxBench: Benchmarking Large Language Models for Multitask Fraud & Abuse Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Anirban Majumder, Dan Ma, Joymallya Chakraborty, Naveed Janvekar, Walid Chaabene, Wei Xia","submitted_at":"2024-09-09T21:12:03Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities in natural language processing tasks. However, their practical application in high-stake domains, such as fraud and abuse detection, remains an area that requires further exploration. The existing applications often narrowly focus on specific tasks like toxicity or hate speech detection. In this paper, we present a comprehensive benchmark suite designed to assess the performance of LLMs in identifying and mitigating fraudulent and abusive language across various real-world scenarios. Our benchmark encompasses a diverse set "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.06072","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-09T21:12:03Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1772386dc62f88e9c46314415f90a35ed1ff2376753655ee1eeaf99aef8154a4","abstract_canon_sha256":"5e84980e783b5c28aea8237efa61ebea5570f43fcfd4a933fe296cc04c8afa02"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:05:01.873099Z","signature_b64":"av0EYDz2LjqWkoWvDNl+F2/v3uOtBrU/VAIj6Y5IC8+RJnziK8zcYmLyxApDrS8g8WqBQHe9417dZVts+H0mCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0b730b0b69632433c31bf3fe39d2a19b55659e67448ce031b3659faf500cf27b","last_reissued_at":"2026-07-05T09:05:01.872650Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:05:01.872650Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DetoxBench: Benchmarking Large Language Models for Multitask Fraud & Abuse Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Anirban Majumder, Dan Ma, Joymallya Chakraborty, Naveed Janvekar, Walid Chaabene, Wei Xia","submitted_at":"2024-09-09T21:12:03Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities in natural language processing tasks. However, their practical application in high-stake domains, such as fraud and abuse detection, remains an area that requires further exploration. The existing applications often narrowly focus on specific tasks like toxicity or hate speech detection. In this paper, we present a comprehensive benchmark suite designed to assess the performance of LLMs in identifying and mitigating fraudulent and abusive language across various real-world scenarios. Our benchmark encompasses a diverse set "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.06072","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.06072/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.06072","created_at":"2026-07-05T09:05:01.872704+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.06072v1","created_at":"2026-07-05T09:05:01.872704+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.06072","created_at":"2026-07-05T09:05:01.872704+00:00"},{"alias_kind":"pith_short_12","alias_value":"BNZQWC3JMMSD","created_at":"2026-07-05T09:05:01.872704+00:00"},{"alias_kind":"pith_short_16","alias_value":"BNZQWC3JMMSDHQY3","created_at":"2026-07-05T09:05:01.872704+00:00"},{"alias_kind":"pith_short_8","alias_value":"BNZQWC3J","created_at":"2026-07-05T09:05:01.872704+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20759","citing_title":"Rethinking Fraud Safety Evaluation: Multi-Round Attacks Reveal Safety-Utility Tradeoffs in Graph-Context LLM Defenders","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BNZQWC3JMMSDHQY36P7DTUVBTN","json":"https://pith.science/pith/BNZQWC3JMMSDHQY36P7DTUVBTN.json","graph_json":"https://pith.science/api/pith-number/BNZQWC3JMMSDHQY36P7DTUVBTN/graph.json","events_json":"https://pith.science/api/pith-number/BNZQWC3JMMSDHQY36P7DTUVBTN/events.json","paper":"https://pith.science/paper/BNZQWC3J"},"agent_actions":{"view_html":"https://pith.science/pith/BNZQWC3JMMSDHQY36P7DTUVBTN","download_json":"https://pith.science/pith/BNZQWC3JMMSDHQY36P7DTUVBTN.json","view_paper":"https://pith.science/paper/BNZQWC3J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.06072&json=true","fetch_graph":"https://pith.science/api/pith-number/BNZQWC3JMMSDHQY36P7DTUVBTN/graph.json","fetch_events":"https://pith.science/api/pith-number/BNZQWC3JMMSDHQY36P7DTUVBTN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BNZQWC3JMMSDHQY36P7DTUVBTN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BNZQWC3JMMSDHQY36P7DTUVBTN/action/storage_attestation","attest_author":"https://pith.science/pith/BNZQWC3JMMSDHQY36P7DTUVBTN/action/author_attestation","sign_citation":"https://pith.science/pith/BNZQWC3JMMSDHQY36P7DTUVBTN/action/citation_signature","submit_replication":"https://pith.science/pith/BNZQWC3JMMSDHQY36P7DTUVBTN/action/replication_record"}},"created_at":"2026-07-05T09:05:01.872704+00:00","updated_at":"2026-07-05T09:05:01.872704+00:00"}