{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AUKRHC7OZXS2TMP2WFXENWHBNA","short_pith_number":"pith:AUKRHC7O","schema_version":"1.0","canonical_sha256":"0515138beecde5a9b1fab16e46d8e168065759abef7a9636f206f61c3c4959a1","source":{"kind":"arxiv","id":"2402.11963","version":1},"attestation_state":"computed","paper":{"title":"Imbalance in Regression Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Kowatsch, Kilian Tscharke, Konstantin B\\\"otinger, Nicolas M. M\\\"uller, Philip Sperl","submitted_at":"2024-02-19T09:06:26Z","abstract_excerpt":"For classification, the problem of class imbalance is well known and has been extensively studied. In this paper, we argue that imbalance in regression is an equally important problem which has so far been overlooked: Due to under- and over-representations in a data set's target distribution, regressors are prone to degenerate to naive models, systematically neglecting uncommon training data and over-representing targets seen often during training. We analyse this problem theoretically and use resulting insights to develop a first definition of imbalance in regression, which we show to be a ge"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.11963","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-19T09:06:26Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a92a3a09e02a713c2456ca2abd45a44c2063909248dc98938a870a8063d5cfa9","abstract_canon_sha256":"87f8709d29cf79db3c0e7c18e466c014328adeefe8a9dce79abc631e7fea0734"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:46:46.709566Z","signature_b64":"F6g/d53/riS4dwyfhjanpStKHXAGtKQ1KJQ71B+tTnzJMEp2nqS36ba34goT3A7nT+XG2Ie2VXmB2DcDajKVDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0515138beecde5a9b1fab16e46d8e168065759abef7a9636f206f61c3c4959a1","last_reissued_at":"2026-07-05T07:46:46.708986Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:46:46.708986Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Imbalance in Regression Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Kowatsch, Kilian Tscharke, Konstantin B\\\"otinger, Nicolas M. M\\\"uller, Philip Sperl","submitted_at":"2024-02-19T09:06:26Z","abstract_excerpt":"For classification, the problem of class imbalance is well known and has been extensively studied. In this paper, we argue that imbalance in regression is an equally important problem which has so far been overlooked: Due to under- and over-representations in a data set's target distribution, regressors are prone to degenerate to naive models, systematically neglecting uncommon training data and over-representing targets seen often during training. We analyse this problem theoretically and use resulting insights to develop a first definition of imbalance in regression, which we show to be a ge"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.11963","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.11963/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.11963","created_at":"2026-07-05T07:46:46.709068+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.11963v1","created_at":"2026-07-05T07:46:46.709068+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.11963","created_at":"2026-07-05T07:46:46.709068+00:00"},{"alias_kind":"pith_short_12","alias_value":"AUKRHC7OZXS2","created_at":"2026-07-05T07:46:46.709068+00:00"},{"alias_kind":"pith_short_16","alias_value":"AUKRHC7OZXS2TMP2","created_at":"2026-07-05T07:46:46.709068+00:00"},{"alias_kind":"pith_short_8","alias_value":"AUKRHC7O","created_at":"2026-07-05T07:46:46.709068+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.01486","citing_title":"Model-agnostic Mitigation Strategies of Data Imbalance for Regression","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AUKRHC7OZXS2TMP2WFXENWHBNA","json":"https://pith.science/pith/AUKRHC7OZXS2TMP2WFXENWHBNA.json","graph_json":"https://pith.science/api/pith-number/AUKRHC7OZXS2TMP2WFXENWHBNA/graph.json","events_json":"https://pith.science/api/pith-number/AUKRHC7OZXS2TMP2WFXENWHBNA/events.json","paper":"https://pith.science/paper/AUKRHC7O"},"agent_actions":{"view_html":"https://pith.science/pith/AUKRHC7OZXS2TMP2WFXENWHBNA","download_json":"https://pith.science/pith/AUKRHC7OZXS2TMP2WFXENWHBNA.json","view_paper":"https://pith.science/paper/AUKRHC7O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.11963&json=true","fetch_graph":"https://pith.science/api/pith-number/AUKRHC7OZXS2TMP2WFXENWHBNA/graph.json","fetch_events":"https://pith.science/api/pith-number/AUKRHC7OZXS2TMP2WFXENWHBNA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AUKRHC7OZXS2TMP2WFXENWHBNA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AUKRHC7OZXS2TMP2WFXENWHBNA/action/storage_attestation","attest_author":"https://pith.science/pith/AUKRHC7OZXS2TMP2WFXENWHBNA/action/author_attestation","sign_citation":"https://pith.science/pith/AUKRHC7OZXS2TMP2WFXENWHBNA/action/citation_signature","submit_replication":"https://pith.science/pith/AUKRHC7OZXS2TMP2WFXENWHBNA/action/replication_record"}},"created_at":"2026-07-05T07:46:46.709068+00:00","updated_at":"2026-07-05T07:46:46.709068+00:00"}