{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:J3TSDHL2ERNYYT22DGYMAZJW5J","short_pith_number":"pith:J3TSDHL2","schema_version":"1.0","canonical_sha256":"4ee7219d7a245b8c4f5a19b0c06536ea4c15168748bfeed91c60f94f6a8355ce","source":{"kind":"arxiv","id":"2412.14527","version":1},"attestation_state":"computed","paper":{"title":"Statistical Undersampling with Mutual Information and Support Points","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Alex Mak, Linglong Kong, Shivani Pandey, Shubham Sahoo, Yidan Yue","submitted_at":"2024-12-19T04:48:29Z","abstract_excerpt":"Class imbalance and distributional differences in large datasets present significant challenges for classification tasks machine learning, often leading to biased models and poor predictive performance for minority classes. This work introduces two novel undersampling approaches: mutual information-based stratified simple random sampling and support points optimization. These methods prioritize representative data selection, effectively minimizing information loss. Empirical results across multiple classification tasks demonstrate that our methods outperform traditional undersampling technique"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.14527","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2024-12-19T04:48:29Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"9b7ac4fa5b918b8c69c8b90fe9dd12c564b1b70ba823d56564aeddee4be7dcd6","abstract_canon_sha256":"6fa22bc0d0c26ff06f12581485e2748f0e8a94c8601e888f634d83f03ee14473"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:51:48.874986Z","signature_b64":"J+EdlwY8xd1nETIiu8zKOqr/p6YT4Vg3JmU32Zqvd0kM8cPKObsw0mtkC4Ef7iBsAMm61b3KaILMZIJXvJJIBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ee7219d7a245b8c4f5a19b0c06536ea4c15168748bfeed91c60f94f6a8355ce","last_reissued_at":"2026-07-05T09:51:48.874365Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:51:48.874365Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Statistical Undersampling with Mutual Information and Support Points","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Alex Mak, Linglong Kong, Shivani Pandey, Shubham Sahoo, Yidan Yue","submitted_at":"2024-12-19T04:48:29Z","abstract_excerpt":"Class imbalance and distributional differences in large datasets present significant challenges for classification tasks machine learning, often leading to biased models and poor predictive performance for minority classes. This work introduces two novel undersampling approaches: mutual information-based stratified simple random sampling and support points optimization. These methods prioritize representative data selection, effectively minimizing information loss. Empirical results across multiple classification tasks demonstrate that our methods outperform traditional undersampling technique"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.14527","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.14527/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.14527","created_at":"2026-07-05T09:51:48.874446+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.14527v1","created_at":"2026-07-05T09:51:48.874446+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.14527","created_at":"2026-07-05T09:51:48.874446+00:00"},{"alias_kind":"pith_short_12","alias_value":"J3TSDHL2ERNY","created_at":"2026-07-05T09:51:48.874446+00:00"},{"alias_kind":"pith_short_16","alias_value":"J3TSDHL2ERNYYT22","created_at":"2026-07-05T09:51:48.874446+00:00"},{"alias_kind":"pith_short_8","alias_value":"J3TSDHL2","created_at":"2026-07-05T09:51:48.874446+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19998","citing_title":"Tri-Info: Generalizable, Interpretable Failure Prediction for VLA Models via Information Theory","ref_index":249,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J3TSDHL2ERNYYT22DGYMAZJW5J","json":"https://pith.science/pith/J3TSDHL2ERNYYT22DGYMAZJW5J.json","graph_json":"https://pith.science/api/pith-number/J3TSDHL2ERNYYT22DGYMAZJW5J/graph.json","events_json":"https://pith.science/api/pith-number/J3TSDHL2ERNYYT22DGYMAZJW5J/events.json","paper":"https://pith.science/paper/J3TSDHL2"},"agent_actions":{"view_html":"https://pith.science/pith/J3TSDHL2ERNYYT22DGYMAZJW5J","download_json":"https://pith.science/pith/J3TSDHL2ERNYYT22DGYMAZJW5J.json","view_paper":"https://pith.science/paper/J3TSDHL2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.14527&json=true","fetch_graph":"https://pith.science/api/pith-number/J3TSDHL2ERNYYT22DGYMAZJW5J/graph.json","fetch_events":"https://pith.science/api/pith-number/J3TSDHL2ERNYYT22DGYMAZJW5J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J3TSDHL2ERNYYT22DGYMAZJW5J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J3TSDHL2ERNYYT22DGYMAZJW5J/action/storage_attestation","attest_author":"https://pith.science/pith/J3TSDHL2ERNYYT22DGYMAZJW5J/action/author_attestation","sign_citation":"https://pith.science/pith/J3TSDHL2ERNYYT22DGYMAZJW5J/action/citation_signature","submit_replication":"https://pith.science/pith/J3TSDHL2ERNYYT22DGYMAZJW5J/action/replication_record"}},"created_at":"2026-07-05T09:51:48.874446+00:00","updated_at":"2026-07-05T09:51:48.874446+00:00"}