{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:NNHSS4VJGBAATGV6FSJL2JVBAS","short_pith_number":"pith:NNHSS4VJ","schema_version":"1.0","canonical_sha256":"6b4f2972a93040099abe2c92bd26a104afa5db728ae2c7eac8093b1124290170","source":{"kind":"arxiv","id":"2208.14417","version":3},"attestation_state":"computed","paper":{"title":"Fraud Dataset Benchmark and Applications","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anqi Cheng, Hao Zhou, Jakub Zablocki, Jianbo Liu, Julia Xu, Justin Tittelfitz, Prince Grover, Zheng Li","submitted_at":"2022-08-30T17:35:39Z","abstract_excerpt":"Standardized datasets and benchmarks have spurred innovations in computer vision, natural language processing, multi-modal and tabular settings. We note that, as compared to other well researched fields, fraud detection has unique challenges: high-class imbalance, diverse feature types, frequently changing fraud patterns, and adversarial nature of the problem. Due to these, the modeling approaches evaluated on datasets from other research fields may not work well for the fraud detection. In this paper, we introduce Fraud Dataset Benchmark (FDB), a compilation of publicly available datasets cat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.14417","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-08-30T17:35:39Z","cross_cats_sorted":["cs.CR","stat.ML"],"title_canon_sha256":"b8432879bcd14edf71a4f9962d6fe152a5ddc5172fb14b795c02e8feb4f8a3d6","abstract_canon_sha256":"aa97cd5006584fd75b369590ae48855b24742fb4680cf0b41a140b54b8864d4d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:53:31.232048Z","signature_b64":"pb3C11oTN8bVA4qsGx2ilLDNNvOInrWWfnDa9V8VMqbRqDeY26uRSGIuWZl10pvIZYe7KkMORt8teglQrIazDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6b4f2972a93040099abe2c92bd26a104afa5db728ae2c7eac8093b1124290170","last_reissued_at":"2026-07-05T06:53:31.231469Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:53:31.231469Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fraud Dataset Benchmark and Applications","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anqi Cheng, Hao Zhou, Jakub Zablocki, Jianbo Liu, Julia Xu, Justin Tittelfitz, Prince Grover, Zheng Li","submitted_at":"2022-08-30T17:35:39Z","abstract_excerpt":"Standardized datasets and benchmarks have spurred innovations in computer vision, natural language processing, multi-modal and tabular settings. We note that, as compared to other well researched fields, fraud detection has unique challenges: high-class imbalance, diverse feature types, frequently changing fraud patterns, and adversarial nature of the problem. Due to these, the modeling approaches evaluated on datasets from other research fields may not work well for the fraud detection. In this paper, we introduce Fraud Dataset Benchmark (FDB), a compilation of publicly available datasets cat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.14417","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.14417/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.14417","created_at":"2026-07-05T06:53:31.231535+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.14417v3","created_at":"2026-07-05T06:53:31.231535+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.14417","created_at":"2026-07-05T06:53:31.231535+00:00"},{"alias_kind":"pith_short_12","alias_value":"NNHSS4VJGBAA","created_at":"2026-07-05T06:53:31.231535+00:00"},{"alias_kind":"pith_short_16","alias_value":"NNHSS4VJGBAATGV6","created_at":"2026-07-05T06:53:31.231535+00:00"},{"alias_kind":"pith_short_8","alias_value":"NNHSS4VJ","created_at":"2026-07-05T06:53:31.231535+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08146","citing_title":"SAGE: An LLM-driven Self Reflective Agentic Framework for Fraud Detection","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26135","citing_title":"SilIF: Silhouette-Augmented Isolation Forest for Unsupervised Transaction Fraud Detection","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28554","citing_title":"High Performance, Low Reliability: Uncertainty Benchmarking for Tabular Foundation Models","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NNHSS4VJGBAATGV6FSJL2JVBAS","json":"https://pith.science/pith/NNHSS4VJGBAATGV6FSJL2JVBAS.json","graph_json":"https://pith.science/api/pith-number/NNHSS4VJGBAATGV6FSJL2JVBAS/graph.json","events_json":"https://pith.science/api/pith-number/NNHSS4VJGBAATGV6FSJL2JVBAS/events.json","paper":"https://pith.science/paper/NNHSS4VJ"},"agent_actions":{"view_html":"https://pith.science/pith/NNHSS4VJGBAATGV6FSJL2JVBAS","download_json":"https://pith.science/pith/NNHSS4VJGBAATGV6FSJL2JVBAS.json","view_paper":"https://pith.science/paper/NNHSS4VJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.14417&json=true","fetch_graph":"https://pith.science/api/pith-number/NNHSS4VJGBAATGV6FSJL2JVBAS/graph.json","fetch_events":"https://pith.science/api/pith-number/NNHSS4VJGBAATGV6FSJL2JVBAS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NNHSS4VJGBAATGV6FSJL2JVBAS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NNHSS4VJGBAATGV6FSJL2JVBAS/action/storage_attestation","attest_author":"https://pith.science/pith/NNHSS4VJGBAATGV6FSJL2JVBAS/action/author_attestation","sign_citation":"https://pith.science/pith/NNHSS4VJGBAATGV6FSJL2JVBAS/action/citation_signature","submit_replication":"https://pith.science/pith/NNHSS4VJGBAATGV6FSJL2JVBAS/action/replication_record"}},"created_at":"2026-07-05T06:53:31.231535+00:00","updated_at":"2026-07-05T06:53:31.231535+00:00"}