{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:J4BIII6NFIFZJWON2MBPIUPHSB","short_pith_number":"pith:J4BIII6N","schema_version":"1.0","canonical_sha256":"4f028423cd2a0b94d9cdd302f451e7907f454234d9c6d845706c68721b0aab4e","source":{"kind":"arxiv","id":"2410.13114","version":1},"attestation_state":"computed","paper":{"title":"Sound Check: Auditing Audio Datasets","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","eess.AS"],"primary_cat":"cs.SD","authors_text":"Annie Chu, Ezra Awumey, Harry H. Jiang, Julia Barnett, Michael Feffer, Rachel Hong, Robin Netzorg, Sauvik Das, William Agnew","submitted_at":"2024-10-17T00:51:27Z","abstract_excerpt":"Generative audio models are rapidly advancing in both capabilities and public utilization -- several powerful generative audio models have readily available open weights, and some tech companies have released high quality generative audio products. Yet, while prior work has enumerated many ethical issues stemming from the data on which generative visual and textual models have been trained, we have little understanding of similar issues with generative audio datasets, including those related to bias, toxicity, and intellectual property. To bridge this gap, we conducted a literature review of h"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.13114","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SD","submitted_at":"2024-10-17T00:51:27Z","cross_cats_sorted":["cs.AI","cs.CY","eess.AS"],"title_canon_sha256":"983a639edfeaa6fc545f287e6c8858db58bcfb2638a585e6372a64339f7c6cd6","abstract_canon_sha256":"ff90a952a46f0cff23825824fcd9f44c876e346eb9018908a45a153385172b8c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:55.630919Z","signature_b64":"Z2YoyW4o4/6d0xaDR3KRYCAg+r85FkI9owkZN+pLFQghsbvhOBiuXzau9gqVI0hTCAVN3MjfDLDzEUiMRvWADA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f028423cd2a0b94d9cdd302f451e7907f454234d9c6d845706c68721b0aab4e","last_reissued_at":"2026-07-05T09:21:55.630374Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:55.630374Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sound Check: Auditing Audio Datasets","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","eess.AS"],"primary_cat":"cs.SD","authors_text":"Annie Chu, Ezra Awumey, Harry H. Jiang, Julia Barnett, Michael Feffer, Rachel Hong, Robin Netzorg, Sauvik Das, William Agnew","submitted_at":"2024-10-17T00:51:27Z","abstract_excerpt":"Generative audio models are rapidly advancing in both capabilities and public utilization -- several powerful generative audio models have readily available open weights, and some tech companies have released high quality generative audio products. Yet, while prior work has enumerated many ethical issues stemming from the data on which generative visual and textual models have been trained, we have little understanding of similar issues with generative audio datasets, including those related to bias, toxicity, and intellectual property. To bridge this gap, we conducted a literature review of h"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.13114","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.13114/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.13114","created_at":"2026-07-05T09:21:55.630446+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.13114v1","created_at":"2026-07-05T09:21:55.630446+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.13114","created_at":"2026-07-05T09:21:55.630446+00:00"},{"alias_kind":"pith_short_12","alias_value":"J4BIII6NFIFZ","created_at":"2026-07-05T09:21:55.630446+00:00"},{"alias_kind":"pith_short_16","alias_value":"J4BIII6NFIFZJWON","created_at":"2026-07-05T09:21:55.630446+00:00"},{"alias_kind":"pith_short_8","alias_value":"J4BIII6N","created_at":"2026-07-05T09:21:55.630446+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.04230","citing_title":"XAttnMark: Learning Robust Audio Watermarking with Cross-Attention","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24794","citing_title":"V.O.I.C.E (Voice, Ownership, Identity, Control, Expression): Risk Taxonomy of Synthetic Voice Generation From Empirical Data","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13776","citing_title":"Who Gets Flagged? The Pluralistic Evaluation Gap in AI Content Watermarking","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J4BIII6NFIFZJWON2MBPIUPHSB","json":"https://pith.science/pith/J4BIII6NFIFZJWON2MBPIUPHSB.json","graph_json":"https://pith.science/api/pith-number/J4BIII6NFIFZJWON2MBPIUPHSB/graph.json","events_json":"https://pith.science/api/pith-number/J4BIII6NFIFZJWON2MBPIUPHSB/events.json","paper":"https://pith.science/paper/J4BIII6N"},"agent_actions":{"view_html":"https://pith.science/pith/J4BIII6NFIFZJWON2MBPIUPHSB","download_json":"https://pith.science/pith/J4BIII6NFIFZJWON2MBPIUPHSB.json","view_paper":"https://pith.science/paper/J4BIII6N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.13114&json=true","fetch_graph":"https://pith.science/api/pith-number/J4BIII6NFIFZJWON2MBPIUPHSB/graph.json","fetch_events":"https://pith.science/api/pith-number/J4BIII6NFIFZJWON2MBPIUPHSB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J4BIII6NFIFZJWON2MBPIUPHSB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J4BIII6NFIFZJWON2MBPIUPHSB/action/storage_attestation","attest_author":"https://pith.science/pith/J4BIII6NFIFZJWON2MBPIUPHSB/action/author_attestation","sign_citation":"https://pith.science/pith/J4BIII6NFIFZJWON2MBPIUPHSB/action/citation_signature","submit_replication":"https://pith.science/pith/J4BIII6NFIFZJWON2MBPIUPHSB/action/replication_record"}},"created_at":"2026-07-05T09:21:55.630446+00:00","updated_at":"2026-07-05T09:21:55.630446+00:00"}