{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JHVJY2HEZUO3FZ3DLKACACIXW2","short_pith_number":"pith:JHVJY2HE","schema_version":"1.0","canonical_sha256":"49ea9c68e4cd1db2e7635a80200917b682ce70d7fc02aa5fac1e9f6e5bd951d6","source":{"kind":"arxiv","id":"2408.14939","version":1},"attestation_state":"computed","paper":{"title":"Integrating Continuous and Binary Relevances in Audio-Text Relevance Learning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Huang Xie, Khazar Khorrami, Okko R\\\"as\\\"anen, Tuomas Virtanen","submitted_at":"2024-08-27T10:23:26Z","abstract_excerpt":"Audio-text relevance learning refers to learning the shared semantic properties of audio samples and textual descriptions. The standard approach uses binary relevances derived from pairs of audio samples and their human-provided captions, categorizing each pair as either positive or negative. This may result in suboptimal systems due to varying levels of relevance between audio samples and captions. In contrast, a recent study used human-assigned relevance ratings, i.e., continuous relevances, for these pairs but did not obtain performance gains in audio-text relevance learning. This work intr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.14939","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"eess.AS","submitted_at":"2024-08-27T10:23:26Z","cross_cats_sorted":[],"title_canon_sha256":"f29a74f05a39ef3cf4597600e9581ceb7073c10a83ccda53882779fbff2ab841","abstract_canon_sha256":"b20396f94720ab7a98e1acf927cde47587f84e8a7b329c8b191320383a41310e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:59:41.033180Z","signature_b64":"yjISPNAzIPaEh/qoQr0J+q2QUQbE4cKvrDS1xKZBys02+8EmoBUrUpfEeh/0B0YVo4vZkIe4+BJaUZDLDCxwAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49ea9c68e4cd1db2e7635a80200917b682ce70d7fc02aa5fac1e9f6e5bd951d6","last_reissued_at":"2026-07-05T08:59:41.032773Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:59:41.032773Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Integrating Continuous and Binary Relevances in Audio-Text Relevance Learning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Huang Xie, Khazar Khorrami, Okko R\\\"as\\\"anen, Tuomas Virtanen","submitted_at":"2024-08-27T10:23:26Z","abstract_excerpt":"Audio-text relevance learning refers to learning the shared semantic properties of audio samples and textual descriptions. The standard approach uses binary relevances derived from pairs of audio samples and their human-provided captions, categorizing each pair as either positive or negative. This may result in suboptimal systems due to varying levels of relevance between audio samples and captions. In contrast, a recent study used human-assigned relevance ratings, i.e., continuous relevances, for these pairs but did not obtain performance gains in audio-text relevance learning. This work intr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.14939","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.14939/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.14939","created_at":"2026-07-05T08:59:41.032828+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.14939v1","created_at":"2026-07-05T08:59:41.032828+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.14939","created_at":"2026-07-05T08:59:41.032828+00:00"},{"alias_kind":"pith_short_12","alias_value":"JHVJY2HEZUO3","created_at":"2026-07-05T08:59:41.032828+00:00"},{"alias_kind":"pith_short_16","alias_value":"JHVJY2HEZUO3FZ3D","created_at":"2026-07-05T08:59:41.032828+00:00"},{"alias_kind":"pith_short_8","alias_value":"JHVJY2HE","created_at":"2026-07-05T08:59:41.032828+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.01356","citing_title":"Text-based Audio Retrieval by Learning from Similarities between Audio Captions","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JHVJY2HEZUO3FZ3DLKACACIXW2","json":"https://pith.science/pith/JHVJY2HEZUO3FZ3DLKACACIXW2.json","graph_json":"https://pith.science/api/pith-number/JHVJY2HEZUO3FZ3DLKACACIXW2/graph.json","events_json":"https://pith.science/api/pith-number/JHVJY2HEZUO3FZ3DLKACACIXW2/events.json","paper":"https://pith.science/paper/JHVJY2HE"},"agent_actions":{"view_html":"https://pith.science/pith/JHVJY2HEZUO3FZ3DLKACACIXW2","download_json":"https://pith.science/pith/JHVJY2HEZUO3FZ3DLKACACIXW2.json","view_paper":"https://pith.science/paper/JHVJY2HE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.14939&json=true","fetch_graph":"https://pith.science/api/pith-number/JHVJY2HEZUO3FZ3DLKACACIXW2/graph.json","fetch_events":"https://pith.science/api/pith-number/JHVJY2HEZUO3FZ3DLKACACIXW2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JHVJY2HEZUO3FZ3DLKACACIXW2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JHVJY2HEZUO3FZ3DLKACACIXW2/action/storage_attestation","attest_author":"https://pith.science/pith/JHVJY2HEZUO3FZ3DLKACACIXW2/action/author_attestation","sign_citation":"https://pith.science/pith/JHVJY2HEZUO3FZ3DLKACACIXW2/action/citation_signature","submit_replication":"https://pith.science/pith/JHVJY2HEZUO3FZ3DLKACACIXW2/action/replication_record"}},"created_at":"2026-07-05T08:59:41.032828+00:00","updated_at":"2026-07-05T08:59:41.032828+00:00"}