{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QPNZLLMCV6SIAXSIYJAXJRCSA3","short_pith_number":"pith:QPNZLLMC","schema_version":"1.0","canonical_sha256":"83db95ad82afa4805e48c24174c45206c6711086f62fe4dc5cbce15bc846503b","source":{"kind":"arxiv","id":"2303.01267","version":1},"attestation_state":"computed","paper":{"title":"Token Contrast for Weakly-Supervised Semantic Segmentation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Du, Heliang Zheng, Lixiang Ru, Yibing Zhan","submitted_at":"2023-03-02T13:51:58Z","abstract_excerpt":"Weakly-Supervised Semantic Segmentation (WSSS) using image-level labels typically utilizes Class Activation Map (CAM) to generate the pseudo labels. Limited by the local structure perception of CNN, CAM usually cannot identify the integral object regions. Though the recent Vision Transformer (ViT) can remedy this flaw, we observe it also brings the over-smoothing issue, \\ie, the final patch tokens incline to be uniform. In this work, we propose Token Contrast (ToCo) to address this issue and further explore the virtue of ViT for WSSS. Firstly, motivated by the observation that intermediate lay"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.01267","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-03-02T13:51:58Z","cross_cats_sorted":[],"title_canon_sha256":"241c03808ac3bb0d49de966da4b91b3f53fa2feb60eee5b7e89c31a20bcd7f48","abstract_canon_sha256":"aeaccad0753250fff7c5b05c905a09c207df9a41915bfd8a85d2febcccd5c594"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:47:28.930026Z","signature_b64":"65YQu94iLBY96zZ71hdJp1s/3i/u3Xa7Ti8Jr3US1QTyeiWa2wdMoHE5jnEvfGBmoYHDxLSsBjcZNrId/vjlAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"83db95ad82afa4805e48c24174c45206c6711086f62fe4dc5cbce15bc846503b","last_reissued_at":"2026-07-05T05:47:28.929616Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:47:28.929616Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Token Contrast for Weakly-Supervised Semantic Segmentation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Du, Heliang Zheng, Lixiang Ru, Yibing Zhan","submitted_at":"2023-03-02T13:51:58Z","abstract_excerpt":"Weakly-Supervised Semantic Segmentation (WSSS) using image-level labels typically utilizes Class Activation Map (CAM) to generate the pseudo labels. Limited by the local structure perception of CNN, CAM usually cannot identify the integral object regions. Though the recent Vision Transformer (ViT) can remedy this flaw, we observe it also brings the over-smoothing issue, \\ie, the final patch tokens incline to be uniform. In this work, we propose Token Contrast (ToCo) to address this issue and further explore the virtue of ViT for WSSS. Firstly, motivated by the observation that intermediate lay"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.01267","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.01267/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.01267","created_at":"2026-07-05T05:47:28.929689+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.01267v1","created_at":"2026-07-05T05:47:28.929689+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.01267","created_at":"2026-07-05T05:47:28.929689+00:00"},{"alias_kind":"pith_short_12","alias_value":"QPNZLLMCV6SI","created_at":"2026-07-05T05:47:28.929689+00:00"},{"alias_kind":"pith_short_16","alias_value":"QPNZLLMCV6SIAXSI","created_at":"2026-07-05T05:47:28.929689+00:00"},{"alias_kind":"pith_short_8","alias_value":"QPNZLLMC","created_at":"2026-07-05T05:47:28.929689+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30577","citing_title":"APRIL-MedSeg: A Modular Medical Image Segmentation Toolbox Embracing Modern Paradigms","ref_index":245,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30577","citing_title":"APRIL-MedSeg: A Modular Medical Image Segmentation Toolbox Embracing Modern Paradigms","ref_index":194,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QPNZLLMCV6SIAXSIYJAXJRCSA3","json":"https://pith.science/pith/QPNZLLMCV6SIAXSIYJAXJRCSA3.json","graph_json":"https://pith.science/api/pith-number/QPNZLLMCV6SIAXSIYJAXJRCSA3/graph.json","events_json":"https://pith.science/api/pith-number/QPNZLLMCV6SIAXSIYJAXJRCSA3/events.json","paper":"https://pith.science/paper/QPNZLLMC"},"agent_actions":{"view_html":"https://pith.science/pith/QPNZLLMCV6SIAXSIYJAXJRCSA3","download_json":"https://pith.science/pith/QPNZLLMCV6SIAXSIYJAXJRCSA3.json","view_paper":"https://pith.science/paper/QPNZLLMC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.01267&json=true","fetch_graph":"https://pith.science/api/pith-number/QPNZLLMCV6SIAXSIYJAXJRCSA3/graph.json","fetch_events":"https://pith.science/api/pith-number/QPNZLLMCV6SIAXSIYJAXJRCSA3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QPNZLLMCV6SIAXSIYJAXJRCSA3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QPNZLLMCV6SIAXSIYJAXJRCSA3/action/storage_attestation","attest_author":"https://pith.science/pith/QPNZLLMCV6SIAXSIYJAXJRCSA3/action/author_attestation","sign_citation":"https://pith.science/pith/QPNZLLMCV6SIAXSIYJAXJRCSA3/action/citation_signature","submit_replication":"https://pith.science/pith/QPNZLLMCV6SIAXSIYJAXJRCSA3/action/replication_record"}},"created_at":"2026-07-05T05:47:28.929689+00:00","updated_at":"2026-07-05T05:47:28.929689+00:00"}