{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:445VZB4I5DEW4Q7G3HY5JZHWIE","short_pith_number":"pith:445VZB4I","schema_version":"1.0","canonical_sha256":"e73b5c8788e8c96e43e6d9f1d4e4f6411a61b881b28cf33d77c9b57f0ddc585e","source":{"kind":"arxiv","id":"2406.02345","version":2},"attestation_state":"computed","paper":{"title":"Progressive Confident Masking Attention Network for Audio-Visual Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Feng Dong, Jinchao Zhu, Shuyue Zhu, Yuxuan Wang","submitted_at":"2024-06-04T14:21:41Z","abstract_excerpt":"Audio and visual signals typically occur simultaneously, and humans possess an innate ability to correlate and synchronize information from these two modalities. Recently, a challenging problem known as Audio-Visual Segmentation (AVS) has emerged, intending to produce segmentation maps for sounding objects within a scene. However, the methods proposed so far have not sufficiently integrated audio and visual information, and the computational costs have been extremely high. Additionally, the outputs of different stages have not been fully utilized. To facilitate this research, we introduce a no"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.02345","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-04T14:21:41Z","cross_cats_sorted":["cs.AI","cs.LG","cs.MM"],"title_canon_sha256":"3c377d1059645b2d041ae9cd991913c9783f427149cb27e301af7d1b583aea0e","abstract_canon_sha256":"bfca9f2c4923e1edbae6d9dbf41d0440b9f3ae36a58184ee1436a20e49ee8553"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:20.551903Z","signature_b64":"/mT+dzdQW6O/syE8mDZPtJ/jemG8oZxz58fw74SDK8mBT+icXEZaNMZgMm76E4RFWFNkVmyVazc2SVPk8qyvCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e73b5c8788e8c96e43e6d9f1d4e4f6411a61b881b28cf33d77c9b57f0ddc585e","last_reissued_at":"2026-07-05T10:11:20.551391Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:20.551391Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Progressive Confident Masking Attention Network for Audio-Visual Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Feng Dong, Jinchao Zhu, Shuyue Zhu, Yuxuan Wang","submitted_at":"2024-06-04T14:21:41Z","abstract_excerpt":"Audio and visual signals typically occur simultaneously, and humans possess an innate ability to correlate and synchronize information from these two modalities. Recently, a challenging problem known as Audio-Visual Segmentation (AVS) has emerged, intending to produce segmentation maps for sounding objects within a scene. However, the methods proposed so far have not sufficiently integrated audio and visual information, and the computational costs have been extremely high. Additionally, the outputs of different stages have not been fully utilized. To facilitate this research, we introduce a no"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.02345","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.02345/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.02345","created_at":"2026-07-05T10:11:20.551452+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.02345v2","created_at":"2026-07-05T10:11:20.551452+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.02345","created_at":"2026-07-05T10:11:20.551452+00:00"},{"alias_kind":"pith_short_12","alias_value":"445VZB4I5DEW","created_at":"2026-07-05T10:11:20.551452+00:00"},{"alias_kind":"pith_short_16","alias_value":"445VZB4I5DEW4Q7G","created_at":"2026-07-05T10:11:20.551452+00:00"},{"alias_kind":"pith_short_8","alias_value":"445VZB4I","created_at":"2026-07-05T10:11:20.551452+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/445VZB4I5DEW4Q7G3HY5JZHWIE","json":"https://pith.science/pith/445VZB4I5DEW4Q7G3HY5JZHWIE.json","graph_json":"https://pith.science/api/pith-number/445VZB4I5DEW4Q7G3HY5JZHWIE/graph.json","events_json":"https://pith.science/api/pith-number/445VZB4I5DEW4Q7G3HY5JZHWIE/events.json","paper":"https://pith.science/paper/445VZB4I"},"agent_actions":{"view_html":"https://pith.science/pith/445VZB4I5DEW4Q7G3HY5JZHWIE","download_json":"https://pith.science/pith/445VZB4I5DEW4Q7G3HY5JZHWIE.json","view_paper":"https://pith.science/paper/445VZB4I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.02345&json=true","fetch_graph":"https://pith.science/api/pith-number/445VZB4I5DEW4Q7G3HY5JZHWIE/graph.json","fetch_events":"https://pith.science/api/pith-number/445VZB4I5DEW4Q7G3HY5JZHWIE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/445VZB4I5DEW4Q7G3HY5JZHWIE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/445VZB4I5DEW4Q7G3HY5JZHWIE/action/storage_attestation","attest_author":"https://pith.science/pith/445VZB4I5DEW4Q7G3HY5JZHWIE/action/author_attestation","sign_citation":"https://pith.science/pith/445VZB4I5DEW4Q7G3HY5JZHWIE/action/citation_signature","submit_replication":"https://pith.science/pith/445VZB4I5DEW4Q7G3HY5JZHWIE/action/replication_record"}},"created_at":"2026-07-05T10:11:20.551452+00:00","updated_at":"2026-07-05T10:11:20.551452+00:00"}