{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LKMMJZYRTKPAD6QVILDXDMR645","short_pith_number":"pith:LKMMJZYR","schema_version":"1.0","canonical_sha256":"5a98c4e7119a9e01fa1542c771b23ee76292342c325be43a5de3db9d585e93e2","source":{"kind":"arxiv","id":"2506.07572","version":1},"attestation_state":"computed","paper":{"title":"Learning Speaker-Invariant Visual Features for Lipreading","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dan Guo, Feng Xue, Jinrui Zhang, Richang Hong, Shuang Yang, Shujie Li, Yu Li","submitted_at":"2025-06-09T09:16:14Z","abstract_excerpt":"Lipreading is a challenging cross-modal task that aims to convert visual lip movements into spoken text. Existing lipreading methods often extract visual features that include speaker-specific lip attributes (e.g., shape, color, texture), which introduce spurious correlations between vision and text. These correlations lead to suboptimal lipreading accuracy and restrict model generalization. To address this challenge, we introduce SIFLip, a speaker-invariant visual feature learning framework that disentangles speaker-specific attributes using two complementary disentanglement modules (Implicit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07572","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-09T09:16:14Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"cefa3146a8754211d354b253bd0a0cd4844d70d74e1330a480786682a99e17b7","abstract_canon_sha256":"f9bde279792a04fac6b3e4f2900d6ff69d0bfaa5f9b4a6f6bae9515bece09a4a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:29.624682Z","signature_b64":"LuuRa9zRecZ9UD7sLrErNl6jP5P1tPu8jtICUYHPjcwsSEt5169kCZRNjr5C48Fxad8Hhz/wiL9MKx81RbCvBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a98c4e7119a9e01fa1542c771b23ee76292342c325be43a5de3db9d585e93e2","last_reissued_at":"2026-07-05T11:18:29.624107Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:29.624107Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Speaker-Invariant Visual Features for Lipreading","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dan Guo, Feng Xue, Jinrui Zhang, Richang Hong, Shuang Yang, Shujie Li, Yu Li","submitted_at":"2025-06-09T09:16:14Z","abstract_excerpt":"Lipreading is a challenging cross-modal task that aims to convert visual lip movements into spoken text. Existing lipreading methods often extract visual features that include speaker-specific lip attributes (e.g., shape, color, texture), which introduce spurious correlations between vision and text. These correlations lead to suboptimal lipreading accuracy and restrict model generalization. To address this challenge, we introduce SIFLip, a speaker-invariant visual feature learning framework that disentangles speaker-specific attributes using two complementary disentanglement modules (Implicit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07572","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07572/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07572","created_at":"2026-07-05T11:18:29.624186+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07572v1","created_at":"2026-07-05T11:18:29.624186+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07572","created_at":"2026-07-05T11:18:29.624186+00:00"},{"alias_kind":"pith_short_12","alias_value":"LKMMJZYRTKPA","created_at":"2026-07-05T11:18:29.624186+00:00"},{"alias_kind":"pith_short_16","alias_value":"LKMMJZYRTKPAD6QV","created_at":"2026-07-05T11:18:29.624186+00:00"},{"alias_kind":"pith_short_8","alias_value":"LKMMJZYR","created_at":"2026-07-05T11:18:29.624186+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LKMMJZYRTKPAD6QVILDXDMR645","json":"https://pith.science/pith/LKMMJZYRTKPAD6QVILDXDMR645.json","graph_json":"https://pith.science/api/pith-number/LKMMJZYRTKPAD6QVILDXDMR645/graph.json","events_json":"https://pith.science/api/pith-number/LKMMJZYRTKPAD6QVILDXDMR645/events.json","paper":"https://pith.science/paper/LKMMJZYR"},"agent_actions":{"view_html":"https://pith.science/pith/LKMMJZYRTKPAD6QVILDXDMR645","download_json":"https://pith.science/pith/LKMMJZYRTKPAD6QVILDXDMR645.json","view_paper":"https://pith.science/paper/LKMMJZYR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07572&json=true","fetch_graph":"https://pith.science/api/pith-number/LKMMJZYRTKPAD6QVILDXDMR645/graph.json","fetch_events":"https://pith.science/api/pith-number/LKMMJZYRTKPAD6QVILDXDMR645/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LKMMJZYRTKPAD6QVILDXDMR645/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LKMMJZYRTKPAD6QVILDXDMR645/action/storage_attestation","attest_author":"https://pith.science/pith/LKMMJZYRTKPAD6QVILDXDMR645/action/author_attestation","sign_citation":"https://pith.science/pith/LKMMJZYRTKPAD6QVILDXDMR645/action/citation_signature","submit_replication":"https://pith.science/pith/LKMMJZYRTKPAD6QVILDXDMR645/action/replication_record"}},"created_at":"2026-07-05T11:18:29.624186+00:00","updated_at":"2026-07-05T11:18:29.624186+00:00"}