{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NLMLN5U3KKXBB4RF5GCLHOCFVQ","short_pith_number":"pith:NLMLN5U3","schema_version":"1.0","canonical_sha256":"6ad8b6f69b52ae10f225e984b3b845ac3789c1bc8a94d9b2e2e87de57286f54b","source":{"kind":"arxiv","id":"2407.10947","version":1},"attestation_state":"computed","paper":{"title":"Can Textual Semantics Mitigate Sounding Object Segmentation Preference?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Di Hu, Honggang Zhang, Peiwen Sun, Yaoting Wang, Yuanchao Li","submitted_at":"2024-07-15T17:45:20Z","abstract_excerpt":"The Audio-Visual Segmentation (AVS) task aims to segment sounding objects in the visual space using audio cues. However, in this work, it is recognized that previous AVS methods show a heavy reliance on detrimental segmentation preferences related to audible objects, rather than precise audio guidance. We argue that the primary reason is that audio lacks robust semantics compared to vision, especially in multi-source sounding scenes, resulting in weak audio guidance over the visual space. Motivated by the the fact that text modality is well explored and contains rich abstract semantics, we pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.10947","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-15T17:45:20Z","cross_cats_sorted":[],"title_canon_sha256":"46d6acd9c92345b95eaed3356fc05846d837450773fa8c72730c7729aaf77cb9","abstract_canon_sha256":"07239fbc541965f1c109e3912d9d776e6e7bf4ff0e267628642ddbc87acc920d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:06.674332Z","signature_b64":"qVFokXdXmtpwWdG6M95q9ZgF8XCyhfsSRx+nbnyDa+emEJZuh98VpXcLrLH6QmZh6ncpadODYFIN4kjFsGLGCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6ad8b6f69b52ae10f225e984b3b845ac3789c1bc8a94d9b2e2e87de57286f54b","last_reissued_at":"2026-07-05T08:44:06.673866Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:06.673866Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Textual Semantics Mitigate Sounding Object Segmentation Preference?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Di Hu, Honggang Zhang, Peiwen Sun, Yaoting Wang, Yuanchao Li","submitted_at":"2024-07-15T17:45:20Z","abstract_excerpt":"The Audio-Visual Segmentation (AVS) task aims to segment sounding objects in the visual space using audio cues. However, in this work, it is recognized that previous AVS methods show a heavy reliance on detrimental segmentation preferences related to audible objects, rather than precise audio guidance. We argue that the primary reason is that audio lacks robust semantics compared to vision, especially in multi-source sounding scenes, resulting in weak audio guidance over the visual space. Motivated by the the fact that text modality is well explored and contains rich abstract semantics, we pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.10947","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.10947/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.10947","created_at":"2026-07-05T08:44:06.673921+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.10947v1","created_at":"2026-07-05T08:44:06.673921+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.10947","created_at":"2026-07-05T08:44:06.673921+00:00"},{"alias_kind":"pith_short_12","alias_value":"NLMLN5U3KKXB","created_at":"2026-07-05T08:44:06.673921+00:00"},{"alias_kind":"pith_short_16","alias_value":"NLMLN5U3KKXBB4RF","created_at":"2026-07-05T08:44:06.673921+00:00"},{"alias_kind":"pith_short_8","alias_value":"NLMLN5U3","created_at":"2026-07-05T08:44:06.673921+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.00358","citing_title":"Do Audio-Visual Segmentation Models Truly Segment Sounding Objects?","ref_index":48,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NLMLN5U3KKXBB4RF5GCLHOCFVQ","json":"https://pith.science/pith/NLMLN5U3KKXBB4RF5GCLHOCFVQ.json","graph_json":"https://pith.science/api/pith-number/NLMLN5U3KKXBB4RF5GCLHOCFVQ/graph.json","events_json":"https://pith.science/api/pith-number/NLMLN5U3KKXBB4RF5GCLHOCFVQ/events.json","paper":"https://pith.science/paper/NLMLN5U3"},"agent_actions":{"view_html":"https://pith.science/pith/NLMLN5U3KKXBB4RF5GCLHOCFVQ","download_json":"https://pith.science/pith/NLMLN5U3KKXBB4RF5GCLHOCFVQ.json","view_paper":"https://pith.science/paper/NLMLN5U3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.10947&json=true","fetch_graph":"https://pith.science/api/pith-number/NLMLN5U3KKXBB4RF5GCLHOCFVQ/graph.json","fetch_events":"https://pith.science/api/pith-number/NLMLN5U3KKXBB4RF5GCLHOCFVQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NLMLN5U3KKXBB4RF5GCLHOCFVQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NLMLN5U3KKXBB4RF5GCLHOCFVQ/action/storage_attestation","attest_author":"https://pith.science/pith/NLMLN5U3KKXBB4RF5GCLHOCFVQ/action/author_attestation","sign_citation":"https://pith.science/pith/NLMLN5U3KKXBB4RF5GCLHOCFVQ/action/citation_signature","submit_replication":"https://pith.science/pith/NLMLN5U3KKXBB4RF5GCLHOCFVQ/action/replication_record"}},"created_at":"2026-07-05T08:44:06.673921+00:00","updated_at":"2026-07-05T08:44:06.673921+00:00"}