{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LACPG73RSNH6OHWEEQJ4W32IK7","short_pith_number":"pith:LACPG73R","schema_version":"1.0","canonical_sha256":"5804f37f71934fe71ec42413cb6f4857ed8c3174b7039ae7b7a2798f07e3a626","source":{"kind":"arxiv","id":"2411.15787","version":1},"attestation_state":"computed","paper":{"title":"Multi-Token Enhancing for Vision Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo-Wen Yin, Ming-Ming Cheng, Yu-Song Hu, Zhong-Yu Li","submitted_at":"2024-11-24T11:33:17Z","abstract_excerpt":"Vision representation learning, especially self-supervised learning, is pivotal for various vision applications. Ensemble learning has also succeeded in enhancing the performance and robustness of the vision models. However, traditional ensemble strategies are impractical for representation learning, especially self-supervised representation learning that requires large-scale datasets and long schedules. This is because they require k times more training and inference computation costs for an ensemble of k models. Differently, we introduce Multi-Token Enhancing (MTE) that extracts multiple aux"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15787","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-24T11:33:17Z","cross_cats_sorted":[],"title_canon_sha256":"2b51e00791597e46feb1c8ca2ad2911cda82faa1e16d231c748e472d180a78cf","abstract_canon_sha256":"011642baad5f90ded08a3a04e14a04e62a622cf4082e14df6587454de495b053"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:41.092409Z","signature_b64":"cRm5FyoirD4P40A11LV0rhQucxEnWaI8wQ8tXwO/RFgzb7PNUbTn8Y32hHYvuyFQ9V1ZJetoFnnyLTKQPhSeBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5804f37f71934fe71ec42413cb6f4857ed8c3174b7039ae7b7a2798f07e3a626","last_reissued_at":"2026-07-05T09:39:41.091966Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:41.091966Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Token Enhancing for Vision Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo-Wen Yin, Ming-Ming Cheng, Yu-Song Hu, Zhong-Yu Li","submitted_at":"2024-11-24T11:33:17Z","abstract_excerpt":"Vision representation learning, especially self-supervised learning, is pivotal for various vision applications. Ensemble learning has also succeeded in enhancing the performance and robustness of the vision models. However, traditional ensemble strategies are impractical for representation learning, especially self-supervised representation learning that requires large-scale datasets and long schedules. This is because they require k times more training and inference computation costs for an ensemble of k models. Differently, we introduce Multi-Token Enhancing (MTE) that extracts multiple aux"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15787","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15787/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15787","created_at":"2026-07-05T09:39:41.092031+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15787v1","created_at":"2026-07-05T09:39:41.092031+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15787","created_at":"2026-07-05T09:39:41.092031+00:00"},{"alias_kind":"pith_short_12","alias_value":"LACPG73RSNH6","created_at":"2026-07-05T09:39:41.092031+00:00"},{"alias_kind":"pith_short_16","alias_value":"LACPG73RSNH6OHWE","created_at":"2026-07-05T09:39:41.092031+00:00"},{"alias_kind":"pith_short_8","alias_value":"LACPG73R","created_at":"2026-07-05T09:39:41.092031+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.03798","citing_title":"Position: Foundation Models Need Digital Twin Representations","ref_index":50,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LACPG73RSNH6OHWEEQJ4W32IK7","json":"https://pith.science/pith/LACPG73RSNH6OHWEEQJ4W32IK7.json","graph_json":"https://pith.science/api/pith-number/LACPG73RSNH6OHWEEQJ4W32IK7/graph.json","events_json":"https://pith.science/api/pith-number/LACPG73RSNH6OHWEEQJ4W32IK7/events.json","paper":"https://pith.science/paper/LACPG73R"},"agent_actions":{"view_html":"https://pith.science/pith/LACPG73RSNH6OHWEEQJ4W32IK7","download_json":"https://pith.science/pith/LACPG73RSNH6OHWEEQJ4W32IK7.json","view_paper":"https://pith.science/paper/LACPG73R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15787&json=true","fetch_graph":"https://pith.science/api/pith-number/LACPG73RSNH6OHWEEQJ4W32IK7/graph.json","fetch_events":"https://pith.science/api/pith-number/LACPG73RSNH6OHWEEQJ4W32IK7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LACPG73RSNH6OHWEEQJ4W32IK7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LACPG73RSNH6OHWEEQJ4W32IK7/action/storage_attestation","attest_author":"https://pith.science/pith/LACPG73RSNH6OHWEEQJ4W32IK7/action/author_attestation","sign_citation":"https://pith.science/pith/LACPG73RSNH6OHWEEQJ4W32IK7/action/citation_signature","submit_replication":"https://pith.science/pith/LACPG73RSNH6OHWEEQJ4W32IK7/action/replication_record"}},"created_at":"2026-07-05T09:39:41.092031+00:00","updated_at":"2026-07-05T09:39:41.092031+00:00"}