{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:C3WCFOQLTJTM7H22AQUG55QVXW","short_pith_number":"pith:C3WCFOQL","schema_version":"1.0","canonical_sha256":"16ec22ba0b9a66cf9f5a04286ef615bd8e56176a537bd5b2fc4499c91f290cca","source":{"kind":"arxiv","id":"2502.08960","version":3},"attestation_state":"computed","paper":{"title":"A Comprehensive Survey on Imbalanced Data Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chong Chen, Conghui He, Dongting Xie, Hongzhi Yin, Wentao Zhang, Xinyi Gao, Yihang Zhang, Zhengren Wang","submitted_at":"2025-02-13T04:53:17Z","abstract_excerpt":"With the expansion of data availability, machine learning (ML) has achieved remarkable breakthroughs in both academia and industry. However, imbalanced data distributions are prevalent in various types of raw data and severely hinder the performance of ML by biasing the decision-making processes. To deepen the understanding of imbalanced data and facilitate the related research and applications, this survey systematically analyzes various real-world data formats and concludes existing researches for different data formats into four distinct categories: data re-balancing, feature representation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.08960","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-13T04:53:17Z","cross_cats_sorted":[],"title_canon_sha256":"6eb9764f38c2006cf1396dc594b05ddb2e4dfa1367966fc1700ea71bb9af5e5d","abstract_canon_sha256":"790bff6b239d3f01e125be3c4d9130b1491a9f710c9708d8b8874e9fc3209f45"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:10:55.645440Z","signature_b64":"6Ii8/oKEkYjE4N7EcBaMzWku+mmcuVryHKSMmpRc7rXQlu9Qa7/66/ESsLYODzK9fOMn9xDrAmX8uFo9Kx4VDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"16ec22ba0b9a66cf9f5a04286ef615bd8e56176a537bd5b2fc4499c91f290cca","last_reissued_at":"2026-07-05T12:10:55.644880Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:10:55.644880Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comprehensive Survey on Imbalanced Data Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chong Chen, Conghui He, Dongting Xie, Hongzhi Yin, Wentao Zhang, Xinyi Gao, Yihang Zhang, Zhengren Wang","submitted_at":"2025-02-13T04:53:17Z","abstract_excerpt":"With the expansion of data availability, machine learning (ML) has achieved remarkable breakthroughs in both academia and industry. However, imbalanced data distributions are prevalent in various types of raw data and severely hinder the performance of ML by biasing the decision-making processes. To deepen the understanding of imbalanced data and facilitate the related research and applications, this survey systematically analyzes various real-world data formats and concludes existing researches for different data formats into four distinct categories: data re-balancing, feature representation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.08960","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.08960/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.08960","created_at":"2026-07-05T12:10:55.644945+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.08960v3","created_at":"2026-07-05T12:10:55.644945+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.08960","created_at":"2026-07-05T12:10:55.644945+00:00"},{"alias_kind":"pith_short_12","alias_value":"C3WCFOQLTJTM","created_at":"2026-07-05T12:10:55.644945+00:00"},{"alias_kind":"pith_short_16","alias_value":"C3WCFOQLTJTM7H22","created_at":"2026-07-05T12:10:55.644945+00:00"},{"alias_kind":"pith_short_8","alias_value":"C3WCFOQL","created_at":"2026-07-05T12:10:55.644945+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.15970","citing_title":"100x Cost & Latency Reduction: Performance Analysis of AI Query Approximation using Lightweight Proxy Models","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C3WCFOQLTJTM7H22AQUG55QVXW","json":"https://pith.science/pith/C3WCFOQLTJTM7H22AQUG55QVXW.json","graph_json":"https://pith.science/api/pith-number/C3WCFOQLTJTM7H22AQUG55QVXW/graph.json","events_json":"https://pith.science/api/pith-number/C3WCFOQLTJTM7H22AQUG55QVXW/events.json","paper":"https://pith.science/paper/C3WCFOQL"},"agent_actions":{"view_html":"https://pith.science/pith/C3WCFOQLTJTM7H22AQUG55QVXW","download_json":"https://pith.science/pith/C3WCFOQLTJTM7H22AQUG55QVXW.json","view_paper":"https://pith.science/paper/C3WCFOQL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.08960&json=true","fetch_graph":"https://pith.science/api/pith-number/C3WCFOQLTJTM7H22AQUG55QVXW/graph.json","fetch_events":"https://pith.science/api/pith-number/C3WCFOQLTJTM7H22AQUG55QVXW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C3WCFOQLTJTM7H22AQUG55QVXW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C3WCFOQLTJTM7H22AQUG55QVXW/action/storage_attestation","attest_author":"https://pith.science/pith/C3WCFOQLTJTM7H22AQUG55QVXW/action/author_attestation","sign_citation":"https://pith.science/pith/C3WCFOQLTJTM7H22AQUG55QVXW/action/citation_signature","submit_replication":"https://pith.science/pith/C3WCFOQLTJTM7H22AQUG55QVXW/action/replication_record"}},"created_at":"2026-07-05T12:10:55.644945+00:00","updated_at":"2026-07-05T12:10:55.644945+00:00"}