{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6Y2O37GJ4UU4NS47ZPZ4IWURHY","short_pith_number":"pith:6Y2O37GJ","schema_version":"1.0","canonical_sha256":"f634edfcc9e529c6cb9fcbf3c45a913e156a52582f5e09ece3797c46ffa49cab","source":{"kind":"arxiv","id":"2405.05012","version":2},"attestation_state":"computed","paper":{"title":"The Entropy Enigma: Success and Failure of Entropy Minimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Matthias Bethge, Ori Press, Ravid Shwartz-Ziv, Yann LeCun","submitted_at":"2024-05-08T12:26:15Z","abstract_excerpt":"Entropy minimization (EM) is frequently used to increase the accuracy of classification models when they're faced with new data at test time. EM is a self-supervised learning method that optimizes classifiers to assign even higher probabilities to their top predicted classes. In this paper, we analyze why EM works when adapting a model for a few steps and why it eventually fails after adapting for many steps. We show that, at first, EM causes the model to embed test images close to training images, thereby increasing model accuracy. After many steps of optimization, EM makes the model embed te"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.05012","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-08T12:26:15Z","cross_cats_sorted":[],"title_canon_sha256":"1af497302ed8f429553291eea820ec9e62396490b95c948e8ed19336b3ba3794","abstract_canon_sha256":"75d31cd49518bd878c01c7c95c4272b9162a59e79cd9974cf825f568fa764d44"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:18:08.933727Z","signature_b64":"QffYNGhWCsOjBKhpmdbVdP279Qf90FlYCNSzEkQb3/1PB0uxBdsw/sruwFfUek2CLZKbD53N5mK+BES/I8lvBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f634edfcc9e529c6cb9fcbf3c45a913e156a52582f5e09ece3797c46ffa49cab","last_reissued_at":"2026-07-05T08:18:08.933265Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:18:08.933265Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Entropy Enigma: Success and Failure of Entropy Minimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Matthias Bethge, Ori Press, Ravid Shwartz-Ziv, Yann LeCun","submitted_at":"2024-05-08T12:26:15Z","abstract_excerpt":"Entropy minimization (EM) is frequently used to increase the accuracy of classification models when they're faced with new data at test time. EM is a self-supervised learning method that optimizes classifiers to assign even higher probabilities to their top predicted classes. In this paper, we analyze why EM works when adapting a model for a few steps and why it eventually fails after adapting for many steps. We show that, at first, EM causes the model to embed test images close to training images, thereby increasing model accuracy. After many steps of optimization, EM makes the model embed te"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.05012","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.05012/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.05012","created_at":"2026-07-05T08:18:08.933323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.05012v2","created_at":"2026-07-05T08:18:08.933323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.05012","created_at":"2026-07-05T08:18:08.933323+00:00"},{"alias_kind":"pith_short_12","alias_value":"6Y2O37GJ4UU4","created_at":"2026-07-05T08:18:08.933323+00:00"},{"alias_kind":"pith_short_16","alias_value":"6Y2O37GJ4UU4NS47","created_at":"2026-07-05T08:18:08.933323+00:00"},{"alias_kind":"pith_short_8","alias_value":"6Y2O37GJ","created_at":"2026-07-05T08:18:08.933323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02339","citing_title":"Entropy Minimization without Model Collapse: Mitigating Prediction Bias in Medical Imaging","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15134","citing_title":"The Unreasonable Effectiveness of Entropy Minimization in LLM Reasoning","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01967","citing_title":"MER-DG: Modality-Entropy Regularization for Multimodal Domain Generalization","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6Y2O37GJ4UU4NS47ZPZ4IWURHY","json":"https://pith.science/pith/6Y2O37GJ4UU4NS47ZPZ4IWURHY.json","graph_json":"https://pith.science/api/pith-number/6Y2O37GJ4UU4NS47ZPZ4IWURHY/graph.json","events_json":"https://pith.science/api/pith-number/6Y2O37GJ4UU4NS47ZPZ4IWURHY/events.json","paper":"https://pith.science/paper/6Y2O37GJ"},"agent_actions":{"view_html":"https://pith.science/pith/6Y2O37GJ4UU4NS47ZPZ4IWURHY","download_json":"https://pith.science/pith/6Y2O37GJ4UU4NS47ZPZ4IWURHY.json","view_paper":"https://pith.science/paper/6Y2O37GJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.05012&json=true","fetch_graph":"https://pith.science/api/pith-number/6Y2O37GJ4UU4NS47ZPZ4IWURHY/graph.json","fetch_events":"https://pith.science/api/pith-number/6Y2O37GJ4UU4NS47ZPZ4IWURHY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6Y2O37GJ4UU4NS47ZPZ4IWURHY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6Y2O37GJ4UU4NS47ZPZ4IWURHY/action/storage_attestation","attest_author":"https://pith.science/pith/6Y2O37GJ4UU4NS47ZPZ4IWURHY/action/author_attestation","sign_citation":"https://pith.science/pith/6Y2O37GJ4UU4NS47ZPZ4IWURHY/action/citation_signature","submit_replication":"https://pith.science/pith/6Y2O37GJ4UU4NS47ZPZ4IWURHY/action/replication_record"}},"created_at":"2026-07-05T08:18:08.933323+00:00","updated_at":"2026-07-05T08:18:08.933323+00:00"}