{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JAZYA5E3FZNCXK6ROVYQZMAALV","short_pith_number":"pith:JAZYA5E3","schema_version":"1.0","canonical_sha256":"483380749b2e5a2babd175710cb0005d70907822e799bc640d8bd90add861da1","source":{"kind":"arxiv","id":"2404.07815","version":2},"attestation_state":"computed","paper":{"title":"Post-Hoc Reversal: Are We Selecting Models Prematurely?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Carlos Guestrin, Mrigank Raman, Rishabh Ranjan, Saurabh Garg, Zachary Lipton","submitted_at":"2024-04-11T14:58:19Z","abstract_excerpt":"Trained models are often composed with post-hoc transforms such as temperature scaling (TS), ensembling and stochastic weight averaging (SWA) to improve performance, robustness, uncertainty estimation, etc. However, such transforms are typically applied only after the base models have already been finalized by standard means. In this paper, we challenge this practice with an extensive empirical study. In particular, we demonstrate a phenomenon that we call post-hoc reversal, where performance trends are reversed after applying post-hoc transforms. This phenomenon is especially prominent in hig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.07815","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-04-11T14:58:19Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"66abc60555658b80f0feeca70a874c979f1d70d7361dcce6e486f025e728314d","abstract_canon_sha256":"8b678b19412d031a529332ccbadfdeb34f4b8d13f3364b6fe97cb29767b55f85"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:31.298140Z","signature_b64":"oyH3F3E1mxcXdv1EJR1jRvqgmftSIOMb8wj2/Kk9Gkf/zBtOvPTbWmzO4fW/V5GMRpUxxgvsT2ADn/pcJQD3AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"483380749b2e5a2babd175710cb0005d70907822e799bc640d8bd90add861da1","last_reissued_at":"2026-07-05T09:15:31.297654Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:31.297654Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Post-Hoc Reversal: Are We Selecting Models Prematurely?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Carlos Guestrin, Mrigank Raman, Rishabh Ranjan, Saurabh Garg, Zachary Lipton","submitted_at":"2024-04-11T14:58:19Z","abstract_excerpt":"Trained models are often composed with post-hoc transforms such as temperature scaling (TS), ensembling and stochastic weight averaging (SWA) to improve performance, robustness, uncertainty estimation, etc. However, such transforms are typically applied only after the base models have already been finalized by standard means. In this paper, we challenge this practice with an extensive empirical study. In particular, we demonstrate a phenomenon that we call post-hoc reversal, where performance trends are reversed after applying post-hoc transforms. This phenomenon is especially prominent in hig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.07815","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.07815/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.07815","created_at":"2026-07-05T09:15:31.297716+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.07815v2","created_at":"2026-07-05T09:15:31.297716+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.07815","created_at":"2026-07-05T09:15:31.297716+00:00"},{"alias_kind":"pith_short_12","alias_value":"JAZYA5E3FZNC","created_at":"2026-07-05T09:15:31.297716+00:00"},{"alias_kind":"pith_short_16","alias_value":"JAZYA5E3FZNCXK6R","created_at":"2026-07-05T09:15:31.297716+00:00"},{"alias_kind":"pith_short_8","alias_value":"JAZYA5E3","created_at":"2026-07-05T09:15:31.297716+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.19195","citing_title":"Rethinking Early Stopping: Refine, Then Calibrate","ref_index":2009,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JAZYA5E3FZNCXK6ROVYQZMAALV","json":"https://pith.science/pith/JAZYA5E3FZNCXK6ROVYQZMAALV.json","graph_json":"https://pith.science/api/pith-number/JAZYA5E3FZNCXK6ROVYQZMAALV/graph.json","events_json":"https://pith.science/api/pith-number/JAZYA5E3FZNCXK6ROVYQZMAALV/events.json","paper":"https://pith.science/paper/JAZYA5E3"},"agent_actions":{"view_html":"https://pith.science/pith/JAZYA5E3FZNCXK6ROVYQZMAALV","download_json":"https://pith.science/pith/JAZYA5E3FZNCXK6ROVYQZMAALV.json","view_paper":"https://pith.science/paper/JAZYA5E3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.07815&json=true","fetch_graph":"https://pith.science/api/pith-number/JAZYA5E3FZNCXK6ROVYQZMAALV/graph.json","fetch_events":"https://pith.science/api/pith-number/JAZYA5E3FZNCXK6ROVYQZMAALV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JAZYA5E3FZNCXK6ROVYQZMAALV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JAZYA5E3FZNCXK6ROVYQZMAALV/action/storage_attestation","attest_author":"https://pith.science/pith/JAZYA5E3FZNCXK6ROVYQZMAALV/action/author_attestation","sign_citation":"https://pith.science/pith/JAZYA5E3FZNCXK6ROVYQZMAALV/action/citation_signature","submit_replication":"https://pith.science/pith/JAZYA5E3FZNCXK6ROVYQZMAALV/action/replication_record"}},"created_at":"2026-07-05T09:15:31.297716+00:00","updated_at":"2026-07-05T09:15:31.297716+00:00"}