{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:D2NGSP6IA4ADKL7SXBZVIWLV32","short_pith_number":"pith:D2NGSP6I","schema_version":"1.0","canonical_sha256":"1e9a693fc80700352ff2b873545975de87b1c573ecb414bb54d535f7107f8bb8","source":{"kind":"arxiv","id":"2507.06848","version":1},"attestation_state":"computed","paper":{"title":"Know Your Attention Maps: Class-specific Token Masking for Weakly Supervised Semantic Segmentation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Damian Borth, Joelle Hanna","submitted_at":"2025-07-09T13:53:34Z","abstract_excerpt":"Weakly Supervised Semantic Segmentation (WSSS) is a challenging problem that has been extensively studied in recent years. Traditional approaches often rely on external modules like Class Activation Maps to highlight regions of interest and generate pseudo segmentation masks. In this work, we propose an end-to-end method that directly utilizes the attention maps learned by a Vision Transformer (ViT) for WSSS. We propose training a sparse ViT with multiple [CLS] tokens (one for each class), using a random masking strategy to promote [CLS] token - class assignment. At inference time, we aggregat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.06848","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-09T13:53:34Z","cross_cats_sorted":[],"title_canon_sha256":"96bfb2f48a0e4c9fa96d00b8af135743904ea444022ae9374f26b29ec3c7b8cc","abstract_canon_sha256":"0e1f812f8b254452cbde95a4e94166b5457c8958f9a6eb18a7206bdbed4f9aaa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:24.103967Z","signature_b64":"skbGJWeb9QkxN/HNFS0y5owg4ie9qeLxvMFClMKi5LIE9axp9VrZtKeYhCOpI9fUxrdT49b69n3QJVrv4AkUDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1e9a693fc80700352ff2b873545975de87b1c573ecb414bb54d535f7107f8bb8","last_reissued_at":"2026-07-05T11:34:24.103514Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:24.103514Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Know Your Attention Maps: Class-specific Token Masking for Weakly Supervised Semantic Segmentation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Damian Borth, Joelle Hanna","submitted_at":"2025-07-09T13:53:34Z","abstract_excerpt":"Weakly Supervised Semantic Segmentation (WSSS) is a challenging problem that has been extensively studied in recent years. Traditional approaches often rely on external modules like Class Activation Maps to highlight regions of interest and generate pseudo segmentation masks. In this work, we propose an end-to-end method that directly utilizes the attention maps learned by a Vision Transformer (ViT) for WSSS. We propose training a sparse ViT with multiple [CLS] tokens (one for each class), using a random masking strategy to promote [CLS] token - class assignment. At inference time, we aggregat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.06848","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.06848/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.06848","created_at":"2026-07-05T11:34:24.103571+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.06848v1","created_at":"2026-07-05T11:34:24.103571+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.06848","created_at":"2026-07-05T11:34:24.103571+00:00"},{"alias_kind":"pith_short_12","alias_value":"D2NGSP6IA4AD","created_at":"2026-07-05T11:34:24.103571+00:00"},{"alias_kind":"pith_short_16","alias_value":"D2NGSP6IA4ADKL7S","created_at":"2026-07-05T11:34:24.103571+00:00"},{"alias_kind":"pith_short_8","alias_value":"D2NGSP6I","created_at":"2026-07-05T11:34:24.103571+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.04593","citing_title":"DiCLIP: Diffusion Model Enhances CLIP's Dense Knowledge for Weakly Supervised Semantic Segmentation","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D2NGSP6IA4ADKL7SXBZVIWLV32","json":"https://pith.science/pith/D2NGSP6IA4ADKL7SXBZVIWLV32.json","graph_json":"https://pith.science/api/pith-number/D2NGSP6IA4ADKL7SXBZVIWLV32/graph.json","events_json":"https://pith.science/api/pith-number/D2NGSP6IA4ADKL7SXBZVIWLV32/events.json","paper":"https://pith.science/paper/D2NGSP6I"},"agent_actions":{"view_html":"https://pith.science/pith/D2NGSP6IA4ADKL7SXBZVIWLV32","download_json":"https://pith.science/pith/D2NGSP6IA4ADKL7SXBZVIWLV32.json","view_paper":"https://pith.science/paper/D2NGSP6I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.06848&json=true","fetch_graph":"https://pith.science/api/pith-number/D2NGSP6IA4ADKL7SXBZVIWLV32/graph.json","fetch_events":"https://pith.science/api/pith-number/D2NGSP6IA4ADKL7SXBZVIWLV32/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D2NGSP6IA4ADKL7SXBZVIWLV32/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D2NGSP6IA4ADKL7SXBZVIWLV32/action/storage_attestation","attest_author":"https://pith.science/pith/D2NGSP6IA4ADKL7SXBZVIWLV32/action/author_attestation","sign_citation":"https://pith.science/pith/D2NGSP6IA4ADKL7SXBZVIWLV32/action/citation_signature","submit_replication":"https://pith.science/pith/D2NGSP6IA4ADKL7SXBZVIWLV32/action/replication_record"}},"created_at":"2026-07-05T11:34:24.103571+00:00","updated_at":"2026-07-05T11:34:24.103571+00:00"}