{"as_of":"2026-08-14T08:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7d85abc5d7ee8db04bccd2bcd5119f32d0d8fce61e6780e8490203d28264b5f2","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T05:44:30.140670Z","state":"measured"},{"denominator":61,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":61,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T14:34:11.253445Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T14:34:12.126271Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"cited_work":{"arxiv_id":"2412.00175","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.00175","snapshot_observed_at":"2026-08-06T14:34:12.126271Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","venue":"cs.CV","work_id":"909d3d0e-ce10-4bc6-916f-c058cb175d4b","year":2024},"citing_paper":{"arxiv_id":"2507.21157","last_updated":"2025-07-24T22:05:52Z","snapshot_observed_at":"2026-08-14T02:53:48.252152Z","submitted_at":"2025-07-24T22:05:52Z","title":"Unmasking Synthetic Realities in Generative AI: A Comprehensive Review of Adversarially Robust Deepfake Detection Systems","version":1},"reference_index":200,"source":"pdf_text","source_observed_at":"2026-08-06T14:34:11.253445Z"},"links":{"cited_paper":"/paper/2412.00175","citing_paper":"/paper/2507.21157"},"observation_digest":"sha256:349155ffd7ff7f42c3b54cfbf7d9a54c7069efd65071588babac8672de484498","observation_id":"cf789b32-7ebb-4b20-a463-9edcb03fb8f5","resolution":{"observed_at":"2026-08-06T14:34:12.131864Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.00175/citation-record","integrity":"/paper/2412.00175/integrity","json":"/paper/2412.00175/citation-record.json","paper":"/paper/2412.00175"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.919623Z","title":"MesoNet: A compact facial video forgery detection network","venue":null,"work_id":"e46cfc95-6915-4250-a079-6309e2a8f503","year":2018},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.902867Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:8e5c1846154acdbda2f5b980093f70fafb5f3c855ab64ce0ba541e956f5d7aaf","observation_id":"0f1a3568-504f-403d-9794-aed617ec33e9","resolution":{"observed_at":"2026-08-12T05:44:30.924180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.908202Z","title":"Lost in translation: Lip- sync deepfake detection from audio-video mismatch","venue":null,"work_id":"813a84db-35fc-430d-9c6f-bdc215ed723b","year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.907182Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:72a6145fea978bc47780b0c615815f4bf06e5403a7a09d75f4184750d30e40f1","observation_id":"37e61685-63f3-40c6-a3e0-c1df0226a21f","resolution":{"observed_at":"2026-08-12T05:44:30.912186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.897220Z","title":"Is synthetic voice detection research going into the right direction? InCVPRW, pages 71–80, 2022","venue":null,"work_id":"672f7386-127e-428e-b86e-da94cb1ef51e","year":2022},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.910553Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:db54afb5ae77d4287458c401e04651d65a6285b21d34e4ced64b1dc0bc72dc42","observation_id":"a5321d64-0dcc-4c1d-ab08-171b41a7d054","resolution":{"observed_at":"2026-08-12T05:44:30.901135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.885476Z","title":"Glitch in the ma- trix: A large scale benchmark for content driven audio-visual forgery detection and localization.Comput","venue":null,"work_id":"817ff526-5008-44db-9bbd-b751c84eb52b","year":2023},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.914197Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:2a62ad8ca1c85eb393636d393d767877cfa4aefaea293cbbc71711735983f503","observation_id":"ca19337e-d4c9-4cd0-a22b-c7f0ef922525","resolution":{"observed_at":"2026-08-12T05:44:30.889636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.873416Z","title":"MARLIN: Masked autoencoder for facial video rep- resentation learning, 2023","venue":null,"work_id":"aa78670a-b05f-4351-8927-ab1b4293f87a","year":2023},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.918272Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:b47c8a9cfb987d917a747b324063a085f69f5fabd6ad847d41d1b80e643392b2","observation_id":"41e03800-cb17-4061-a8fe-5123ee75ae57","resolution":{"observed_at":"2026-08-12T05:44:30.877434Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.863020Z","title":"A V-Deepfake1M: A large-scale LLM-driven audio-visual deepfake dataset, 2024","venue":null,"work_id":"a6c5edeb-5b94-4164-9830-f21264b748aa","year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.921956Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:34de110ddfcd0a830580b7bef61048a1a3659074d72651a45bf9c35bf94e741c","observation_id":"204c936b-d5a6-4257-aab1-e31ce11da211","resolution":{"observed_at":"2026-08-12T05:44:30.866718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.851329Z","title":null,"venue":null,"work_id":"1b3f6953-ac78-4b7d-bec1-1a1cff057d0f","year":2022},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.925819Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:88aba824a20f9d0d727828998bd748b1ae51325bd7fadb9686d9fbc41c3a48f4","observation_id":"e0d3c3d5-aa07-4aaa-b7a0-db4cddaee990","resolution":{"observed_at":"2026-08-12T05:44:30.854772Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.840080Z","title":"What makes fake images detectable? understanding prop- erties that generalize","venue":null,"work_id":"401df650-7d82-422a-9c76-d8f307a43fdc","year":2020},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.929916Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:10f2e2947659e29c89430cc5163b82eb6e54a69be00c21a2e72912b177b6a758","observation_id":"2f537494-10f5-446d-96c9-ef6ace1da91e","resolution":{"observed_at":"2026-08-12T05:44:30.844018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.828372Z","title":"Xception: Deep learning with depthwise separable convolutions","venue":null,"work_id":"fd7fc401-0cfb-4c7e-aaa1-257f7bb2fd7e","year":2017},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.933869Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:86829f5327880f3c5778c99f6beb5dc8c3c75496a4b5327aa46a2300ff6e7fda","observation_id":"ddcda2a4-a8a4-4a14-b71d-20ec75219149","resolution":{"observed_at":"2026-08-12T05:44:30.832907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.816529Z","title":"Not made for each other-audio- visual dissonance-based deepfake detection and localization","venue":null,"work_id":"5ad33f0a-796b-4100-af67-64dc834fec72","year":2020},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.937924Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:9f6793306bcb55bdf9ee6771b61601f05f7f8a4fff19291aa02534bb19147814","observation_id":"601c765b-d82a-4590-b1fb-204f6be653ca","resolution":{"observed_at":"2026-08-12T05:44:30.820744Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.803816Z","title":null,"venue":null,"work_id":"af9a33e5-f83b-4f9e-934d-a49d85699408","year":2018},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.941961Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:5baaabca0c49b1c4a4bfb7e03d2dac7f3698a7142f38f0016b1c15d9162a468e","observation_id":"02b1325f-7e4f-4c1d-9e0e-c2c915a5755e","resolution":{"observed_at":"2026-08-12T05:44:30.807771Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.789669Z","title":"Combining EfficientNet and vision transformers for video deepfake detection","venue":null,"work_id":"6b318d7a-6711-4d94-bd8d-05683754c1a7","year":2022},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.946772Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:64f56e679de0bdbecdcfd36c758f22e273707b82aab5d68b3c08b7f0cb49a74e","observation_id":"7522d123-ddf9-45af-8cad-ccd1a55ba0c7","resolution":{"observed_at":"2026-08-12T05:44:30.794869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.774353Z","title":"Raising the bar of AI-generated image detection with CLIP","venue":null,"work_id":"51fc152a-6a49-4ee7-80ec-ceb9676beff7","year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.951523Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:f29a5ed5278f56dc634f99757d0d881010930d5c78c26352883e1f2c77e14ef2","observation_id":"0e91a96a-2aea-46f4-a7b1-e1588ed3383b","resolution":{"observed_at":"2026-08-12T05:44:30.778987Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.762420Z","title":"Zero-shot detection of AI-generated im- ages","venue":null,"work_id":"1778fdc9-8738-4c49-9884-0465ff94acdd","year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.955266Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:beacf884257e5640b003908bd0daf1125308b05985b545f272d04d4cdde1ac7d","observation_id":"f59e3463-fe24-48a2-ac7d-ac2400aaa6f9","resolution":{"observed_at":"2026-08-12T05:44:30.766713Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.750468Z","title":"Real time speech enhancement in the waveform domain","venue":null,"work_id":"90825f50-ef93-42d7-9f5f-a8367ea2349c","year":2020},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.958972Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:e34b1be19a3179e930b6792442017d561d45b30bddce141fc7df7281ecaf863c","observation_id":"4ff1bd2d-e13c-4e45-8d4b-bfbcf5838d64","resolution":{"observed_at":"2026-08-12T05:44:30.754457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.07397","last_updated":"2020-10-28T03:48:28Z","snapshot_observed_at":"2026-07-06T09:28:36.954658Z","submitted_at":"2020-06-12T18:15:55Z","title":"The DeepFake Detection Challenge (DFDC) Dataset","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.07397","snapshot_observed_at":"2026-08-12T05:44:29.962790Z","title":"The deepfake detection challenge (dfdc) dataset.arXiv preprint arXiv:2006.07397, 2020","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.962790Z"},"links":{"cited_paper":"/paper/2006.07397","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:f992320cd4a95a40ebec9f4b424bbada69466aab0f2209dc63a6bf99c654b7cf","observation_id":"9b9bf00c-b21a-4975-b554-beb863f04c45","resolution":{"observed_at":"2026-08-12T05:44:29.962790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.738870Z","title":"Self- supervised video forensics by audio-visual anomaly detec- tion","venue":null,"work_id":"4ad4d4e3-3366-4139-9300-c83bb3f3f87e","year":2023},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.967173Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:63594f667e5af2190c468ad7a5aee86cca0badc49f34796f21573e368c9e6026","observation_id":"b9dc861f-40ff-4064-b2db-040a5d505a29","resolution":{"observed_at":"2026-08-12T05:44:30.742861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.726375Z","title":"Lips don’t lie: A generalisable and robust approach to face forgery detection","venue":null,"work_id":"c5591263-068c-414b-b11f-fcffd03cf774","year":2021},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.970864Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:9008631d14cba626bcf3f91a10b34b997876b6861cd336b1207650f1359bae14","observation_id":"94eb28c8-6cf5-4d2f-b66a-9e4fe7874ada","resolution":{"observed_at":"2026-08-12T05:44:30.730819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.715164Z","title":"Leveraging real talking faces via self- supervision for robust forgery detection","venue":null,"work_id":"67d59434-3adf-4d32-bb20-726d47b10919","year":2022},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.975148Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:cbd525f26f06e96dc53ba35a1c17c30c983d8174df83926956249538fdd0e1dd","observation_id":"61dc8e57-70cd-4bca-b411-02c32fc4731d","resolution":{"observed_at":"2026-08-12T05:44:30.719091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13103","last_updated":"2025-07-06T11:50:10Z","snapshot_observed_at":"2026-08-13T05:45:04.882505Z","submitted_at":"2023-10-19T19:01:26Z","title":"AVTENet: A Human-Cognition-Inspired Audio-Visual Transformer-Based Ensemble Network for Video Deepfake Detection","version":2},"cited_work":{"arxiv_id":"2310.13103","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.13103","snapshot_observed_at":"2026-08-12T05:44:30.363492Z","title":"AVTENet: A Human-Cognition-Inspired Audio-Visual Transformer-Based Ensemble Network for Video Deepfake Detection","venue":"cs.CV","work_id":"38ad3a8c-ab5c-4938-914e-4efb3a34d968","year":2023},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.979061Z"},"links":{"cited_paper":"/paper/2310.13103","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:25a81eb0dc94d3b331cd30040432c2ac7e89e1824a782ca801c59751918d0280","observation_id":"85c606d9-92f6-44ac-964a-1ca897a6c2f6","resolution":{"observed_at":"2026-08-12T05:44:30.367926Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.704504Z","title":"Implicit identity driven deepfake face swapping detection","venue":null,"work_id":"113534eb-3105-4253-bdbd-abdc9854dba2","year":2023},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.983103Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:674c72f8ee568f14052d379240b53eae41069aa5c19aa704b1726d4f35bf41f8","observation_id":"193d8633-31df-4c3c-a4a6-51e11fdb8b32","resolution":{"observed_at":"2026-08-12T05:44:30.708387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.693179Z","title":"Weiss, Quan Wang, Jonathan Shen, Fei Ren, Zhifeng Chen, Patrick Nguyen, Ruoming Pang, Ig- nacio L ´opez-Moreno, and Yonghui Wu","venue":null,"work_id":"498426b1-d9be-4b1e-b00f-ad319d1ef515","year":2018},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.986480Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:2dedcc4e7af958c8308cbd538baa5c95ade0b32e87f6916a4bd0ff8756cbf2a7","observation_id":"ac5c7317-76a3-4300-8343-73765124b811","resolution":{"observed_at":"2026-08-12T05:44:30.697002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.681498Z","title":null,"venue":null,"work_id":"25c2513e-b02d-4b7a-b818-ff4968101e1f","year":2021},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.989809Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:4eae5a65e0f7acaf3c7c790599aaa767b238b97e11ddfb7ed6a30ffee66fa6cf","observation_id":"3b7b43e2-c716-49d5-b0ac-c0690dd3e926","resolution":{"observed_at":"2026-08-12T05:44:30.685497Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.670695Z","title":"Conditional variational autoencoder with adversarial learning for end-to- end text-to-speech","venue":null,"work_id":"050eb199-8c5c-4fb6-9cf6-c645680d049d","year":2021},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.993621Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:ab2d26cf6fc261f639246dee02574948b72d7569ee0f22524ec4a16adfaa05ff","observation_id":"9ee132c4-b704-4c2e-95d3-8b8d019f0cdc","resolution":{"observed_at":"2026-08-12T05:44:30.674615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1812.08685","last_updated":"2018-12-20T16:36:39Z","snapshot_observed_at":"2026-08-09T14:52:48.078687Z","submitted_at":"2018-12-20T16:36:39Z","title":"DeepFakes: a New Threat to Face Recognition? Assessment and Detection","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1812.08685","snapshot_observed_at":"2026-08-12T05:44:29.997426Z","title":"DeepFakes: A new threat to face recognition? Assessment and detection.CoRR, abs/1812.08685, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:29.997426Z"},"links":{"cited_paper":"/paper/1812.08685","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:d519c4218c3a1168dbc780aa8c01bbbe9ef3d3315183508f9f683d953287bed9","observation_id":"6f07a43b-8e35-4e9d-b9c3-d0c55d9f97cd","resolution":{"observed_at":"2026-08-12T05:44:29.997426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.658597Z","title":"Fast face-swap using convolutional neural networks","venue":null,"work_id":"808ba637-0607-4237-808d-1ae5fe3a169b","year":2017},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.001658Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:aceab9ca7f5adedd8ac73ffd017ea613d61dcd2d6b585430ca919d90fadcb8e6","observation_id":"07a55832-9392-4694-8f79-7ab5cd4afac2","resolution":{"observed_at":"2026-08-12T05:44:30.662715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10193","last_updated":"2025-04-11T06:56:20Z","snapshot_observed_at":"2026-08-13T05:02:38.613626Z","submitted_at":"2024-11-15T13:47:33Z","title":"DiMoDif: Discourse Modality-information Differentiation for Audio-visual Deepfake Detection and Localization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10193","snapshot_observed_at":"2026-08-12T05:44:30.005614Z","title":"DiMoDif: Dis- course modality-information differentiation for audio-visual deepfake detection and localization.CoRR, abs/2411.10193,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.005614Z"},"links":{"cited_paper":"/paper/2411.10193","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:d80f21479433f3cb5dda44dfec17553d320aec689651ddadb78b06abc2ce7723","observation_id":"2f3969ed-538d-4505-85f6-cf0859757533","resolution":{"observed_at":"2026-08-12T05:44:30.005614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.646199Z","title":"KoDF: A large-scale Korean deepfake detection dataset","venue":null,"work_id":"ad1e0bf4-6015-4887-8104-04594a62b085","year":null},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.010154Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:f3e609916c87eded6976c394399edf6ee688fe5c9a2c2ae8c98f13e1a4937534","observation_id":"d83fdd33-7043-40ff-bab6-37a75629e3b0","resolution":{"observed_at":"2026-08-12T05:44:30.650538Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07854","last_updated":"2024-06-12T04:06:56Z","snapshot_observed_at":"2026-08-12T23:44:58.967636Z","submitted_at":"2024-06-12T04:06:56Z","title":"Zero-Shot Fake Video Detection by Audio-Visual Consistency","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07854","snapshot_observed_at":"2026-08-12T05:44:30.015336Z","title":"Zero-shot fake video detection by audio-visual consistency.CoRR, abs/2406.07854, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.015336Z"},"links":{"cited_paper":"/paper/2406.07854","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:baa1f751afb20cf64940d9adc3f40defb09204e5dab215f39f3db7a8d7e64316","observation_id":"6f7b9195-4c79-4c78-a07e-383e73b100e3","resolution":{"observed_at":"2026-08-12T05:44:30.015336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.634057Z","title":"SpeechForensics: Audio-visual speech representation learn- ing for face forgery detection","venue":null,"work_id":"4ab44a0c-c563-4439-930f-772193cdf836","year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.019867Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:463703debe2a0caa02465dbd67e9e9287a11002ed11ddbbef7972d1b5f337cac","observation_id":"d103036f-1b70-4766-8d34-16be070831ac","resolution":{"observed_at":"2026-08-12T05:44:30.638655Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.622395Z","title":"Lips are lying: Spotting the temporal inconsistency between audio and visual in lip- syncing deepfakes","venue":null,"work_id":"45c6b20a-d6dd-4287-9258-ab2c43745f6f","year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.024385Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:49f4020d27c4c7a51590c4f868f38741a3c251bcc51ebdd1dd4f27541168f32d","observation_id":"864ae0bf-3730-4f9a-a9dc-ef63542c0bf0","resolution":{"observed_at":"2026-08-12T05:44:30.626444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10736","last_updated":"2024-07-15T14:01:35Z","snapshot_observed_at":"2026-08-12T23:22:01.248869Z","submitted_at":"2024-07-15T14:01:35Z","title":"When Synthetic Traces Hide Real Content: Analysis of Stable Diffusion Image Laundering","version":1},"cited_work":{"arxiv_id":"2407.10736","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.10736","snapshot_observed_at":"2026-08-12T05:44:30.312320Z","title":"When Synthetic Traces Hide Real Content: Analysis of Stable Diffusion Image Laundering","venue":"cs.CV","work_id":"ba44d099-3b95-4717-bbef-d6e2024ef396","year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.028583Z"},"links":{"cited_paper":"/paper/2407.10736","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:01fa3cd523f10134b7873dbff3f35ffccc9c92855865da1d19556ef9ed839504","observation_id":"68f3aa07-355d-4ca7-b494-0495de107945","resolution":{"observed_at":"2026-08-12T05:44:30.318622Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11566","last_updated":"2024-10-04T09:39:44Z","snapshot_observed_at":"2026-08-12T23:21:15.814986Z","submitted_at":"2024-07-16T10:19:14Z","title":"TGIF: Text-Guided Inpainting Forgery Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.11566","snapshot_observed_at":"2026-08-12T05:44:30.032962Z","title":"TGIF: Text-guided inpainting forgery dataset.CoRR, abs/2407.11566, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.032962Z"},"links":{"cited_paper":"/paper/2407.11566","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:12e8cf1da0bb30011018fb8785f77da1826f4ec06692495406f75e9abc2f8c54","observation_id":"2953a0a3-8d77-4e60-acc9-6b1b3fe83f5c","resolution":{"observed_at":"2026-08-12T05:44:30.032962Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.610376Z","title":"Do GANs leave artificial fingerprints? In MIPR, pages 506–511, 2019","venue":null,"work_id":"b39ee349-63d9-4a7d-a695-c1086c52301c","year":2019},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.037292Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:e2f4c8758eb23181e9793cd0d18940259aea216d8f435a7ce78d5208eeb0316b","observation_id":"1fa776b0-864a-4663-a75b-a74cbc5bda9e","resolution":{"observed_at":"2026-08-12T05:44:30.614526Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.12914","last_updated":"2021-09-28T09:06:33Z","snapshot_observed_at":"2026-08-13T18:58:45.544887Z","submitted_at":"2021-06-23T08:28:59Z","title":"Speech is Silver, Silence is Golden: What do ASVspoof-trained Models Really Learn?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.12914","snapshot_observed_at":"2026-08-12T05:44:30.041044Z","title":"M ¨uller, Franziska Dieckmann, Pavel Czempin, Roman Canals, and Konstantin B ¨ottinger","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.041044Z"},"links":{"cited_paper":"/paper/2106.12914","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:9789aa9f9f0775949af94e058a09b546abe55d953030f71cc08fea8036102ea4","observation_id":"5ff8c039-2b88-415d-81d0-c3259f4fd4a5","resolution":{"observed_at":"2026-08-12T05:44:30.041044Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.598735Z","title":"M ¨uller, Piotr Kawa, Wei Herng Choong, Edres- son Casanova, Eren G ¨olge, Thorsten M ¨uller, Piotr Syga, Philip Sperl, and Konstantin B¨ottinger","venue":null,"work_id":"7eeff043-91c4-42af-ac8c-edc88124f807","year":null},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.045120Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:90ff7a2ba82b9b7d972552451ab52b4292e52efbef25b1493899f13894591020","observation_id":"aeb44556-7661-46de-a740-a292c5702c81","resolution":{"observed_at":"2026-08-12T05:44:30.602844Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.586698Z","title":"FSGAN: Subject agnostic face swapping and reenactment","venue":null,"work_id":"de2bd454-8bbb-4859-8469-602c3ecff878","year":2019},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.048924Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:588cb04092b9a80cf714ee5ce46e9262e15c6ed2979e329dba09b9821d6535b7","observation_id":"969af691-c1d9-44d7-a802-6dfc7f92a78c","resolution":{"observed_at":"2026-08-12T05:44:30.591140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.573969Z","title":"Towards uni- versal fake image detectors that generalize across generative models","venue":null,"work_id":"f1b52038-7504-47e8-80e9-0b2bb27e99e8","year":2023},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.052119Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:9e8407afab86fd0c7d239055a0b74530cc0cf283869bb5f82453d0206de5bbc0","observation_id":"ea3a47d5-9301-44c7-a21c-da8c69fd17b4","resolution":{"observed_at":"2026-08-12T05:44:30.578222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.562602Z","title":"A VFF: Audio-visual feature fusion for video deepfake detection","venue":null,"work_id":"45c7a989-9743-45a5-a511-c09d7d0a14e2","year":null},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.055580Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:8034c3e1d60b8f5eec74d5cfde44a95fd29f387e923108439ead249c819e8cd8","observation_id":"5ae1e3f7-bfde-4e9b-90d6-bd3990cf174d","resolution":{"observed_at":"2026-08-12T05:44:30.566161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.552737Z","title":"Towards generalisable and cali- brated audio deepfake detection with self-supervised repre- sentations","venue":null,"work_id":"03f1c488-06d4-424f-9273-602c38cec9e1","year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.059228Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:919e916e205e1129484eed9be4f0b2ade750928594330a9a8202cd3b25b36d05","observation_id":"9b46e0a9-8ec3-41f8-83bc-25fd6f4fbfd0","resolution":{"observed_at":"2026-08-12T05:44:30.556304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.542576Z","title":"Training-free deepfake voice recognition by leveraging large-scale pre-trained models","venue":null,"work_id":"8f043263-77c7-4f9e-9c15-04c4863d5a02","year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.062350Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:f355d489a8f34c2f85498aa780df07d0c9f82b5fea79e1897920384915383510","observation_id":"e781906e-45f3-4e1c-a5e2-6ad8ba2870cb","resolution":{"observed_at":"2026-08-12T05:44:30.546248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.531075Z","title":"Nambood- iri, and C.V","venue":null,"work_id":"f274016d-b415-4bef-9fbd-69d195a42f6a","year":2020},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.065597Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:95ba99214c492ebc918c14cfdfc5606c64588e8ad14ae6adc6e017a18bf6b0ef","observation_id":"b548f3d9-54c6-406e-862c-fca0b7e40564","resolution":{"observed_at":"2026-08-12T05:44:30.535050Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11835","last_updated":"2025-02-26T18:55:40Z","snapshot_observed_at":"2026-08-12T22:22:29.230312Z","submitted_at":"2024-10-15T17:58:07Z","title":"Aligned Datasets Improve Detection of Latent Diffusion-Generated Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11835","snapshot_observed_at":"2026-08-12T05:44:30.069372Z","title":"On the effectiveness of dataset alignment for fake image detection.CoRR, abs/2410.11835, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.069372Z"},"links":{"cited_paper":"/paper/2410.11835","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:8b893ad19df3a9d5f95487c4afb4fa3606133e74ca8434dc1c6b59618711fa7a","observation_id":"a5b3b9be-cb06-4225-af1c-6ac3e4675cc4","resolution":{"observed_at":"2026-08-12T05:44:30.069372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.01458","last_updated":"2023-11-02T17:59:31Z","snapshot_observed_at":"2026-08-13T05:33:56.798179Z","submitted_at":"2023-11-02T17:59:31Z","title":"Detecting Deepfakes Without Seeing Any","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.01458","snapshot_observed_at":"2026-08-12T05:44:30.073469Z","title":"Detecting deep- fakes without seeing any.CoRR, abs/2311.01458, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.073469Z"},"links":{"cited_paper":"/paper/2311.01458","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:cec0f196a0f3236afe7ac9921b1269223bcd8464c4990496d7a172d630355e56","observation_id":"80a80d7d-94d0-4ed7-a609-0174cd2ae0c4","resolution":{"observed_at":"2026-08-12T05:44:30.073469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.519642Z","title":"AEROB- LADE: Training-free detection of latent diffusion images us- ing autoencoder reconstruction error","venue":null,"work_id":"2b1331a5-6bca-4925-8f82-cf391d490285","year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.077820Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:acba854f28d2892bc99701490f74dc979a5a04fe8a8fd3ae99c821e6a5fb1073","observation_id":"ae187751-f545-462d-ba06-4a0eaca1bf26","resolution":{"observed_at":"2026-08-12T05:44:30.523986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.508110Z","title":"FaceForen- sics++: Learning to detect manipulated facial images","venue":null,"work_id":"3a5622c5-c54a-4bd7-9eb0-d5f4440c3b15","year":2019},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.081672Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:f16d3855f35a22334a35e9714fc5349f91b84a9250d3313e0c1046f07038c3a9","observation_id":"d02d19e7-b2af-40a2-8a8e-11703e729397","resolution":{"observed_at":"2026-08-12T05:44:30.512076Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.497161Z","title":"Hosler, Paolo Bestagini, Matthew C","venue":null,"work_id":"ba0f18ae-bf65-41d3-af44-4d4469245df9","year":2023},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.085554Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:58e1118dfe73dce700129076e6d635a29a2fe1dec92f73261a2ec81ee8c7877f","observation_id":"cf3cfcc4-06bd-4920-9546-450aed995371","resolution":{"observed_at":"2026-08-12T05:44:30.500830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.089207Z","title":"A V-Lip-Sync+: Lever- aging A V-HuBERT to exploit multimodal inconsistency for video deepfake detection.CoRR, abs/2311.02733, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.089207Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:3a6a75fa4941dad16ebe2e11a3677fd94857dab96f12cb9d84b96ebb91a59a0a","observation_id":"15fcc1e4-fd8d-4958-a636-fb21c09db0b8","resolution":{"observed_at":"2026-08-12T05:44:30.089207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.485643Z","title":"Learning audio-visual speech representation by masked multimodal cluster prediction","venue":null,"work_id":"bd3b4c79-c270-464b-b3eb-aabbe3f69550","year":2022},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.093019Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:cd4e176a41a109a7064b0417b4fde0bc4c6bea3dbfc69c79ecc806bb4d464c5d","observation_id":"6134641f-d05b-46e0-a138-e99f04eec00d","resolution":{"observed_at":"2026-08-12T05:44:30.489852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.472753Z","title":"Detecting deep- fakes with self-blended images","venue":null,"work_id":"75f2c6f2-7e50-44ec-a986-4d74693db89a","year":2022},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.096651Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:1c12d5908412a4a1c280abaaa983f693d802be0594f2ae0f9b4377482ad74053","observation_id":"1c02577e-a80d-48f7-8c25-d8e8650742b4","resolution":{"observed_at":"2026-08-12T05:44:30.477035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.08849","last_updated":"2024-12-10T15:35:31Z","snapshot_observed_at":"2026-08-12T22:46:03.233054Z","submitted_at":"2024-09-12T17:59:08Z","title":"DeCLIP: Decoding CLIP representations for deepfake localization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.08849","snapshot_observed_at":"2026-08-12T05:44:30.100607Z","title":"DeCLIP: Decoding CLIP representations for deepfake localization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.100607Z"},"links":{"cited_paper":"/paper/2409.08849","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:064872e7f00f37bf2e7880febae292f8cba4e17a80ceac4c038a79e978d9436b","observation_id":"1b0066b4-d6ec-4b9f-a7d7-67e4c79f2bb8","resolution":{"observed_at":"2026-08-12T05:44:30.100607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.462011Z","title":"Lip reading sentences in the wild","venue":null,"work_id":"4c741ac8-c68c-48ab-b29d-f96213c6ccfd","year":2017},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.105280Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:3b85924d974c22f6bf1273e3024ffc4d8a219afaf2779d7950023f47df8c5c2d","observation_id":"92b24c84-6659-4cbd-88c0-6b6ce0746084","resolution":{"observed_at":"2026-08-12T05:44:30.465517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.450234Z","title":"End-to-end anti-spoofing with RawNet2","venue":null,"work_id":"fb97a98b-3378-4c34-bb6f-7bd91d56f333","year":null},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.109954Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:5404962f7d09fe83756a966433e6258ede39ce23fcb50f742823856a65123bd5","observation_id":"870152d8-6ac1-4255-b6bf-b0ff5c6fe1bc","resolution":{"observed_at":"2026-08-12T05:44:30.454182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03748","last_updated":"2019-01-22T18:47:12Z","snapshot_observed_at":"2026-07-06T06:49:24.960992Z","submitted_at":"2018-07-10T16:52:11Z","title":"Representation Learning with Contrastive Predictive Coding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1807.03748","snapshot_observed_at":"2026-08-12T05:44:30.114598Z","title":"Repre- sentation learning with contrastive predictive coding.CoRR, abs/1807.03748, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.114598Z"},"links":{"cited_paper":"/paper/1807.03748","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:dc03e130f6f74dab11e283d24a43dbd0f33a0380a608a6914b6a1be4a20f7d61","observation_id":"d415a12d-aedf-435d-92bd-0b82b23a53da","resolution":{"observed_at":"2026-08-12T05:44:30.114598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.437838Z","title":"Tan, and Haizhou Li","venue":null,"work_id":"0bde8b8f-3f63-4dca-acfe-07ab39e9d046","year":2023},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.119711Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:ab49430a67cf5daaa6def629be77892ca0aace3377ad3f9e7ae8e76e817cced6","observation_id":"54636c10-493f-46fe-8399-0cff0a423dc3","resolution":{"observed_at":"2026-08-12T05:44:30.442071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.425526Z","title":null,"venue":null,"work_id":"0234ce0f-4d44-4cf1-a177-eead9865ca58","year":2019},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.124304Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:8e2f4ee12686f886f0350da3a374dc8ca5e39e4246111eb18ab331325dc470f9","observation_id":"1747dde7-7503-4e95-b232-e51c5471ec7d","resolution":{"observed_at":"2026-08-12T05:44:30.429625Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13495","last_updated":"2024-10-31T09:11:37Z","snapshot_observed_at":"2026-08-12T23:39:16.028605Z","submitted_at":"2024-06-19T12:35:02Z","title":"DF40: Toward Next-Generation Deepfake Detection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13495","snapshot_observed_at":"2026-08-12T05:44:30.128411Z","title":"DF40: Toward next-generation deepfake detection.CoRR, abs/2406.13495,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.128411Z"},"links":{"cited_paper":"/paper/2406.13495","citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:dd9a4daa0799d7d54da7f6ccfc536190b9b01cc0e7bb7d439478cc23ef7697aa","observation_id":"b02d6ff3-e91c-46b0-bbe0-df84d750d801","resolution":{"observed_at":"2026-08-12T05:44:30.128411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.413017Z","title":"10 A V oiD-DF: Audio-visual joint learning for detecting deep- fake.IEEE Trans","venue":null,"work_id":"1f2e0c20-a647-4ebc-bfce-a526f52fe9ff","year":2015},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.132785Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:a12dfd1c51ee33318541ac4b40a3e79035eeec225d3672bfb6f4262114c6b0be","observation_id":"3fd4eb5b-b339-40f2-9821-dadb6a384bd9","resolution":{"observed_at":"2026-08-12T05:44:30.417412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.400905Z","title":"Attributing fake images to GANs: Learning and analyzing GAN fingerprints","venue":null,"work_id":"96b33c5b-28dd-473d-9a4d-0b31dc08d8d6","year":2019},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.136494Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:87e64c6eba7a342d4cd3e1ce6e56a683876c54131ff1a55205c66e5d8add5be1","observation_id":"01786df4-5813-4367-b56e-a9c5706684c4","resolution":{"observed_at":"2026-08-12T05:44:30.405087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:44:30.387737Z","title":"Exploring temporal coherence for more general video face forgery detection","venue":null,"work_id":"86042f58-c8ed-4050-bb5f-ae46fd9e5a3f","year":null},"citing_paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-12T05:44:30.140670Z"},"links":{"citing_paper":"/paper/2412.00175"},"observation_digest":"sha256:7870fc1c21ca2cf69d96d0970a8db4b86a65fbcd4e4cff50942d91ebea6bbf77","observation_id":"56102646-6931-4131-878f-42212b6c5254","resolution":{"observed_at":"2026-08-12T05:44:30.392580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.00175","last_updated":"2025-05-29T14:44:11Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T20:40:20.763413Z","submitted_at":"2024-11-29T18:58:20Z","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":2,"verified_fuzzy":42},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 1 inbound Pith citation observation for arXiv:2412.00175."}