{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6JWNNC3SYLJWTJ7RQOHKGEABET","short_pith_number":"pith:6JWNNC3S","schema_version":"1.0","canonical_sha256":"f26cd68b72c2d369a7f1838ea3100124d2dbe1b5ab3e889f2fcae249fe11d4c2","source":{"kind":"arxiv","id":"2109.13425","version":1},"attestation_state":"computed","paper":{"title":"The JHU submission to VoxSRC-21: Track 3","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Jejin Cho, Jesus Villalba, Najim Dehak","submitted_at":"2021-09-28T01:30:10Z","abstract_excerpt":"This technical report describes Johns Hopkins University speaker recognition system submitted to Voxceleb Speaker Recognition Challenge 2021 Track 3: Self-supervised speaker verification (closed). Our overall training process is similar to the proposed one from the first place team in the last year's VoxSRC2020 challenge. The main difference is a recently proposed non-contrastive self-supervised method in computer vision (CV), distillation with no labels (DINO), is used to train our initial model, which outperformed the last year's contrastive learning based on momentum contrast (MoCo). Also, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.13425","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2021-09-28T01:30:10Z","cross_cats_sorted":["cs.LG","cs.SD"],"title_canon_sha256":"ddc793f7b4b942682a43fbb2cf32f4bc631da56eea52bea6d13627f499d31d05","abstract_canon_sha256":"01fb27cb30c452a3df29a97e3c3e54831c64f1db44a028c1a32f7791c7e2b37e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:18:02.581900Z","signature_b64":"GkxYG6xtpmxMCZqqZ/eKGq1xBYyBYMCeKTXZqOCDYM/KwbZ9OustHPmwTJDEo8+Mx37FIIu0EuU0SCGPJAlJCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f26cd68b72c2d369a7f1838ea3100124d2dbe1b5ab3e889f2fcae249fe11d4c2","last_reissued_at":"2026-07-05T03:18:02.581553Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:18:02.581553Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The JHU submission to VoxSRC-21: Track 3","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Jejin Cho, Jesus Villalba, Najim Dehak","submitted_at":"2021-09-28T01:30:10Z","abstract_excerpt":"This technical report describes Johns Hopkins University speaker recognition system submitted to Voxceleb Speaker Recognition Challenge 2021 Track 3: Self-supervised speaker verification (closed). Our overall training process is similar to the proposed one from the first place team in the last year's VoxSRC2020 challenge. The main difference is a recently proposed non-contrastive self-supervised method in computer vision (CV), distillation with no labels (DINO), is used to train our initial model, which outperformed the last year's contrastive learning based on momentum contrast (MoCo). Also, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.13425","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.13425/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.13425","created_at":"2026-07-05T03:18:02.581608+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.13425v1","created_at":"2026-07-05T03:18:02.581608+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.13425","created_at":"2026-07-05T03:18:02.581608+00:00"},{"alias_kind":"pith_short_12","alias_value":"6JWNNC3SYLJW","created_at":"2026-07-05T03:18:02.581608+00:00"},{"alias_kind":"pith_short_16","alias_value":"6JWNNC3SYLJWTJ7R","created_at":"2026-07-05T03:18:02.581608+00:00"},{"alias_kind":"pith_short_8","alias_value":"6JWNNC3S","created_at":"2026-07-05T03:18:02.581608+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.04147","citing_title":"Enhancing Self-Supervised Speaker Verification Using Similarity-Connected Graphs and GCN","ref_index":25,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6JWNNC3SYLJWTJ7RQOHKGEABET","json":"https://pith.science/pith/6JWNNC3SYLJWTJ7RQOHKGEABET.json","graph_json":"https://pith.science/api/pith-number/6JWNNC3SYLJWTJ7RQOHKGEABET/graph.json","events_json":"https://pith.science/api/pith-number/6JWNNC3SYLJWTJ7RQOHKGEABET/events.json","paper":"https://pith.science/paper/6JWNNC3S"},"agent_actions":{"view_html":"https://pith.science/pith/6JWNNC3SYLJWTJ7RQOHKGEABET","download_json":"https://pith.science/pith/6JWNNC3SYLJWTJ7RQOHKGEABET.json","view_paper":"https://pith.science/paper/6JWNNC3S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.13425&json=true","fetch_graph":"https://pith.science/api/pith-number/6JWNNC3SYLJWTJ7RQOHKGEABET/graph.json","fetch_events":"https://pith.science/api/pith-number/6JWNNC3SYLJWTJ7RQOHKGEABET/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6JWNNC3SYLJWTJ7RQOHKGEABET/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6JWNNC3SYLJWTJ7RQOHKGEABET/action/storage_attestation","attest_author":"https://pith.science/pith/6JWNNC3SYLJWTJ7RQOHKGEABET/action/author_attestation","sign_citation":"https://pith.science/pith/6JWNNC3SYLJWTJ7RQOHKGEABET/action/citation_signature","submit_replication":"https://pith.science/pith/6JWNNC3SYLJWTJ7RQOHKGEABET/action/replication_record"}},"created_at":"2026-07-05T03:18:02.581608+00:00","updated_at":"2026-07-05T03:18:02.581608+00:00"}