{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:S62KEHVN6QKVRS5BVRQXDI4R5O","short_pith_number":"pith:S62KEHVN","schema_version":"1.0","canonical_sha256":"97b4a21eadf41558cba1ac6171a391ebafc38320f2099a80732a10d649813041","source":{"kind":"arxiv","id":"2307.03133","version":1},"attestation_state":"computed","paper":{"title":"Benchmarking Test-Time Adaptation against Distribution Shifts in Image Classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Jian Liang, Lijun Sheng, Ran He, Yongcan Yu","submitted_at":"2023-07-06T16:59:53Z","abstract_excerpt":"Test-time adaptation (TTA) is a technique aimed at enhancing the generalization performance of models by leveraging unlabeled samples solely during prediction. Given the need for robustness in neural network systems when faced with distribution shifts, numerous TTA methods have recently been proposed. However, evaluating these methods is often done under different settings, such as varying distribution shifts, backbones, and designing scenarios, leading to a lack of consistent and fair benchmarks to validate their effectiveness. To address this issue, we present a benchmark that systematically"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.03133","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-06T16:59:53Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"1f2cf01a16b6897077c057136f821f4e4544e3692e1f49682c370bd1152bfb0d","abstract_canon_sha256":"fd287e1ae075eaeaa0bc9423d39a7cb47404f680a007d42d9e59c2f0b97ec843"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:28:28.389323Z","signature_b64":"DJXScPD7iFY+bcDw3QlH4tDsJ3XJSAa7pOefjJzz6TnCzc5bz9dW9zNSVAbErZFaP7BXfHjuofQC5ohHlecPAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"97b4a21eadf41558cba1ac6171a391ebafc38320f2099a80732a10d649813041","last_reissued_at":"2026-07-05T06:28:28.388827Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:28:28.388827Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Test-Time Adaptation against Distribution Shifts in Image Classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Jian Liang, Lijun Sheng, Ran He, Yongcan Yu","submitted_at":"2023-07-06T16:59:53Z","abstract_excerpt":"Test-time adaptation (TTA) is a technique aimed at enhancing the generalization performance of models by leveraging unlabeled samples solely during prediction. Given the need for robustness in neural network systems when faced with distribution shifts, numerous TTA methods have recently been proposed. However, evaluating these methods is often done under different settings, such as varying distribution shifts, backbones, and designing scenarios, leading to a lack of consistent and fair benchmarks to validate their effectiveness. To address this issue, we present a benchmark that systematically"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.03133","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.03133/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.03133","created_at":"2026-07-05T06:28:28.388890+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.03133v1","created_at":"2026-07-05T06:28:28.388890+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.03133","created_at":"2026-07-05T06:28:28.388890+00:00"},{"alias_kind":"pith_short_12","alias_value":"S62KEHVN6QKV","created_at":"2026-07-05T06:28:28.388890+00:00"},{"alias_kind":"pith_short_16","alias_value":"S62KEHVN6QKVRS5B","created_at":"2026-07-05T06:28:28.388890+00:00"},{"alias_kind":"pith_short_8","alias_value":"S62KEHVN","created_at":"2026-07-05T06:28:28.388890+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02339","citing_title":"Entropy Minimization without Model Collapse: Mitigating Prediction Bias in Medical Imaging","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21327","citing_title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S62KEHVN6QKVRS5BVRQXDI4R5O","json":"https://pith.science/pith/S62KEHVN6QKVRS5BVRQXDI4R5O.json","graph_json":"https://pith.science/api/pith-number/S62KEHVN6QKVRS5BVRQXDI4R5O/graph.json","events_json":"https://pith.science/api/pith-number/S62KEHVN6QKVRS5BVRQXDI4R5O/events.json","paper":"https://pith.science/paper/S62KEHVN"},"agent_actions":{"view_html":"https://pith.science/pith/S62KEHVN6QKVRS5BVRQXDI4R5O","download_json":"https://pith.science/pith/S62KEHVN6QKVRS5BVRQXDI4R5O.json","view_paper":"https://pith.science/paper/S62KEHVN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.03133&json=true","fetch_graph":"https://pith.science/api/pith-number/S62KEHVN6QKVRS5BVRQXDI4R5O/graph.json","fetch_events":"https://pith.science/api/pith-number/S62KEHVN6QKVRS5BVRQXDI4R5O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S62KEHVN6QKVRS5BVRQXDI4R5O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S62KEHVN6QKVRS5BVRQXDI4R5O/action/storage_attestation","attest_author":"https://pith.science/pith/S62KEHVN6QKVRS5BVRQXDI4R5O/action/author_attestation","sign_citation":"https://pith.science/pith/S62KEHVN6QKVRS5BVRQXDI4R5O/action/citation_signature","submit_replication":"https://pith.science/pith/S62KEHVN6QKVRS5BVRQXDI4R5O/action/replication_record"}},"created_at":"2026-07-05T06:28:28.388890+00:00","updated_at":"2026-07-05T06:28:28.388890+00:00"}