{"as_of":"2026-08-07T15:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a1cc4029f406e13ca3ba203981afc6ee603ab94a7bb81eb423cfe8e18bd2381c","coverage":[{"denominator":26,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T01:05:45.841972Z","state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-02T05:52:55.818877Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T05:56:39.910208Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"cited_work":{"arxiv_id":"2506.12222","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.12222","snapshot_observed_at":"2026-07-02T05:56:39.910208Z","title":"Sslam: Enhancing self-supervised models with audio mixtures for polyphonic soundscapes","venue":null,"work_id":"37ad0f61-a107-470e-8ddb-24629620e685","year":2025},"citing_paper":{"arxiv_id":"2605.18409","last_updated":"2026-05-18T13:48:52Z","snapshot_observed_at":"2026-07-06T23:29:20.279470Z","submitted_at":"2026-05-18T13:48:52Z","title":"EnvTriCascade: An Environment-Aware Tri-Stage Cascaded Framework for ESDD2 2026 Challenge","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-19T23:45:03.154037Z"},"links":{"cited_paper":"/paper/2506.12222","citing_paper":"/paper/2605.18409"},"observation_digest":"sha256:53af1daf2143b37ebef6bf67971adf629d422eefad3675c5183e87c75c897ece","observation_id":"d3e344d7-7269-4e10-b44c-5fcf151a06f9","resolution":{"observed_at":"2026-05-19T23:47:52.074583Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"cited_work":{"arxiv_id":"2506.12222","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.12222","snapshot_observed_at":"2026-07-02T05:56:39.910208Z","title":"Sslam: Enhancing self-supervised models with audio mixtures for polyphonic soundscapes","venue":null,"work_id":"37ad0f61-a107-470e-8ddb-24629620e685","year":2025},"citing_paper":{"arxiv_id":"2607.00387","last_updated":"2026-07-01T03:32:08Z","snapshot_observed_at":"2026-08-01T10:46:33.437238Z","submitted_at":"2026-07-01T03:32:08Z","title":"From Objectives to Applications: Aligning Architectural Biases in Audio Self-Supervised Learning","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-07-02T05:52:55.818877Z"},"links":{"cited_paper":"/paper/2506.12222","citing_paper":"/paper/2607.00387"},"observation_digest":"sha256:023a4f2e80cb3e97d31bdd9d85ca249790064a855eab5d86f0323a7a29c20fdf","observation_id":"2fc140d0-23bd-4548-9c8a-8b924dd0d98c","resolution":{"observed_at":"2026-07-02T05:56:39.911902Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.12222/citation-record","integrity":"/paper/2506.12222/integrity","json":"/paper/2506.12222/citation-record.json","paper":"/paper/2506.12222"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:05:46.489143Z","title":"In our initial experiments, we observed that this approach yielded worse performance compared to SSLAM on the AS-20K benchmark (39.9 mAP vs","venue":null,"work_id":"093c8127-55fa-480e-b38e-f06f6950993b","year":2025},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:45.841972Z"},"links":{"citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:58b51743a73a962b11193cc53554af919a4adbd9664048fec672faee86b757b9","observation_id":"4184e5d6-aaaa-4143-8a3c-35a947f6d657","resolution":{"observed_at":"2026-08-07T01:05:46.536287Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:05:46.632879Z","title":null,"venue":null,"work_id":"3e212dd1-0e1a-4c30-b6a0-d658dfac3353","year":2025},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:45.741916Z"},"links":{"citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:6ec027898d4a3754c354cf30920e59c8596d547d5684d9fee734dc6b515e2ec3","observation_id":"f6bfd562-bbc0-4c35-a419-d4457273e968","resolution":{"observed_at":"2026-08-07T01:05:46.767689Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.05069","last_updated":"2022-03-29T12:25:02Z","snapshot_observed_at":"2026-08-04T12:25:56.734801Z","submitted_at":"2021-10-11T08:07:50Z","title":"Efficient Training of Audio Transformers with Patchout","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.05069","snapshot_observed_at":"2026-08-07T01:05:44.433337Z","title":"Efficient training of audio Transformers with patchout","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.433337Z"},"links":{"cited_paper":"/paper/2110.05069","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:e1ad36c84e079f2fe45a71be8db7348862dcfc72494b979e2a805e914a1ba83d","observation_id":"c4f98f10-dac5-48fe-9f54-7f3e2be380a6","resolution":{"observed_at":"2026-08-07T01:05:44.433337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1608.03983","last_updated":"2017-05-03T16:28:09Z","snapshot_observed_at":"2026-07-06T05:06:55.589962Z","submitted_at":"2016-08-13T13:46:05Z","title":"SGDR: Stochastic Gradient Descent with Warm Restarts","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1608.03983","snapshot_observed_at":"2026-08-07T01:05:44.507719Z","title":"SGDR: Stochastic gradient descent with warm restarts","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.507719Z"},"links":{"cited_paper":"/paper/1608.03983","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:a6fdf5662a58afdc9fa10d634be72e5574c99c402e801798b10b91c62340613f","observation_id":"0fcc80e2-feda-4148-88eb-e0baa5622bb2","resolution":{"observed_at":"2026-08-07T01:05:44.507719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-07T01:05:44.590534Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.590534Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:edbcee1a61eb2e4b6c1b902e7551c5c17fcdb209c6625617904f94a496b702a7","observation_id":"2e3f4533-e133-4720-834f-0c0df286f6dc","resolution":{"observed_at":"2026-08-07T01:05:44.590534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18799","last_updated":"2024-09-09T16:00:04Z","snapshot_observed_at":"2026-07-06T16:55:13.127567Z","submitted_at":"2023-11-30T18:43:51Z","title":"X-InstructBLIP: A Framework for aligning X-Modal instruction-aware representations to LLMs and Emergent Cross-modal Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18799","snapshot_observed_at":"2026-08-07T01:05:44.701187Z","title":"X-instructblip: A framework for aligning x-modal instruction-aware representations to llms and emergent cross-modal reasoning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.701187Z"},"links":{"cited_paper":"/paper/2311.18799","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:60e65d1580eb75f7baff7038f763a0604bb1beea3aea3997b16e4e0a3f7e63cc","observation_id":"191b7aef-7ad7-4a45-8d33-809fc4b8aa07","resolution":{"observed_at":"2026-08-07T01:05:44.701187Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.08779","last_updated":"2019-12-03T18:19:07Z","snapshot_observed_at":"2026-08-04T08:35:21.943076Z","submitted_at":"2019-04-18T17:53:38Z","title":"SpecAugment: A Simple Data Augmentation Method for Automatic Speech Recognition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.08779","snapshot_observed_at":"2026-08-07T01:05:44.764826Z","title":"Specaugment: A simple data augmentation method for automatic speech recognition","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.764826Z"},"links":{"cited_paper":"/paper/1904.08779","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:02b7989b8c754a8bf4a9c5254f74ede04ddc07dc646c15d1b29515b6bf4ee42b","observation_id":"85231792-3c9a-4dfc-b166-7954c14aa931","resolution":{"observed_at":"2026-08-07T01:05:44.764826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-07T01:05:45.173925Z","title":"Video-llama: An instruction-tuned audio-visual language model for video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:45.173925Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:56f7eeefcd2a1426f3139746fbf45efc2921c3a2ea0c173e69df9e55f307e64e","observation_id":"964ad6b8-1c77-4194-bd9c-710b25a01b72","resolution":{"observed_at":"2026-08-07T01:05:45.173925Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1710.09412","last_updated":"2018-04-27T21:39:25Z","snapshot_observed_at":"2026-07-06T06:06:04.571585Z","submitted_at":"2017-10-25T18:30:49Z","title":"mixup: Beyond Empirical Risk Minimization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1710.09412","snapshot_observed_at":"2026-08-07T01:05:45.237444Z","title":"mixup: Beyond empirical risk minimization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:45.237444Z"},"links":{"cited_paper":"/paper/1710.09412","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:6966ebdf3e9d06691512d3214312d19c55d25565ba316def527471084163f493","observation_id":"e5832718-e9b9-496d-b2ae-a1ff8d684bc3","resolution":{"observed_at":"2026-08-07T01:05:45.237444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.16103","last_updated":"2023-05-25T14:34:08Z","snapshot_observed_at":"2026-07-06T15:33:23.984280Z","submitted_at":"2023-05-25T14:34:08Z","title":"ChatBridge: Bridging Modalities with Large Language Model as a Language Catalyst","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.16103","snapshot_observed_at":"2026-08-07T01:05:45.325056Z","title":"Chatbridge: Bridging modalities with large language model as a language catalyst","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:45.325056Z"},"links":{"cited_paper":"/paper/2305.16103","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:7352ba711b33a385f98dd02e1ab16a376e21e466139e912c0312bfdba3726e20","observation_id":"424e8316-0769-489b-a34f-997025e07cd3","resolution":{"observed_at":"2026-08-07T01:05:45.325056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:05:47.551394Z","title":null,"venue":null,"work_id":"5519f1fc-e90f-4f03-97a2-31c2c4ed3944","year":2025},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:45.373288Z"},"links":{"citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:033e030e6ef9b833ae1c99d6ef0456903cb01d1ee96fbe2245c84c08a7df3643","observation_id":"c89dbd5a-c2c2-4157-838f-f2f7eae8348f","resolution":{"observed_at":"2026-08-07T01:05:47.683309Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:05:47.325041Z","title":null,"venue":null,"work_id":"a244e8f2-b07b-4cfa-94d8-8f121f8def51","year":2025},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:45.491708Z"},"links":{"citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:6b8850d58c99e147a55a97ab4267cec5b22d70f937902937f91801c25342e990","observation_id":"a23beb56-6b39-4f66-944c-119197902a80","resolution":{"observed_at":"2026-08-07T01:05:47.449606Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:05:47.153914Z","title":"Each recording is annotated with a single class","venue":null,"work_id":"19cb16a3-c221-477f-b3ab-2cc09ebba0da","year":2022},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:45.570518Z"},"links":{"citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:549e1dcf5ea3dcd5077d7b3f7163d48221d171a20e422f4777239807582b6baf","observation_id":"6fe2436a-4f32-4d0d-8e0f-2730d18735e1","resolution":{"observed_at":"2026-08-07T01:05:47.187219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:05:46.903081Z","title":"multi-label","venue":null,"work_id":"d91ddc8d-f085-4626-bb1f-7a736dd6380e","year":2025},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:45.645235Z"},"links":{"citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:6ef61c3398a19332b5d6bcc5fcae1662d209985f3b4505dc486ec226b9346b11","observation_id":"39447296-8e86-4078-9738-54bc04c8d365","resolution":{"observed_at":"2026-08-07T01:05:47.004569Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1804.03209","last_updated":"2018-04-09T19:58:17Z","snapshot_observed_at":"2026-07-06T06:32:32.083176Z","submitted_at":"2018-04-09T19:58:17Z","title":"Speech Commands: A Dataset for Limited-Vocabulary Speech Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1804.03209","snapshot_observed_at":"2026-08-07T01:05:45.096815Z","title":"Speech commands: A dataset for limited-vocabulary speech recognition","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":2005,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:45.096815Z"},"links":{"cited_paper":"/paper/1804.03209","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:87fa0444efb221851dbde6d8d80826a4761597ed412638dc23bba9caf2b7adb4","observation_id":"828e21e4-3bb1-4b55-8a3a-0590664e0a7e","resolution":{"observed_at":"2026-08-07T01:05:45.096815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:05:44.906560Z","title":"Dropout: a simple way to prevent neural networks from overfitting","venue":null,"work_id":null,"year":1929},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":2011,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.906560Z"},"links":{"citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:1236a5850dfc323674649e6031857de1ac880a7bb0b6c1a7d712f18711168b1e","observation_id":"807d1e59-79a6-4c6b-a0d5-dc87f44db323","resolution":{"observed_at":"2026-08-07T01:05:44.906560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13289","last_updated":"2024-04-08T06:12:52Z","snapshot_observed_at":"2026-08-02T21:51:36.809095Z","submitted_at":"2023-10-20T05:41:57Z","title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.13289","snapshot_observed_at":"2026-08-07T01:05:45.009230Z","title":"Salmonn: Towards generic hearing abilities for large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:45.009230Z"},"links":{"cited_paper":"/paper/2310.13289","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:740f50a31fb91cb48dc0a108bd32dba9345f21fbf007d015c6bd90927a0a3ae5","observation_id":"08a10811-8325-40d1-b1bd-37c2c3582a20","resolution":{"observed_at":"2026-08-07T01:05:45.009230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:05:47.823265Z","title":"Scaper: A library for soundscape synthesis and augmentation","venue":null,"work_id":"3c7aa55a-ec7c-489d-97a3-17e3d0896b63","year":2017},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.843781Z"},"links":{"citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:1291f2c01f7aa0b16f2e8d17ca832c12d910bf7ccd5fc528aa8b9c9e2e03b0ca","observation_id":"d54b3c52-6a9c-42b0-9fd1-4079138a73a4","resolution":{"observed_at":"2026-08-07T01:05:47.893714Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.06405","last_updated":"2023-01-12T05:22:31Z","snapshot_observed_at":"2026-07-06T13:30:58.601457Z","submitted_at":"2022-07-13T17:59:55Z","title":"Masked Autoencoders that Listen","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.06405","snapshot_observed_at":"2026-08-07T01:05:44.339061Z","title":"Masked autoencoders that listen","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.339061Z"},"links":{"cited_paper":"/paper/2207.06405","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:2fac82df1d99e802e65726de50649b55cd949c350b5c93f03bfe489e2014a99e","observation_id":"aa434d91-4f9b-4397-8fd1-5e44eef43cdd","resolution":{"observed_at":"2026-08-07T01:05:44.339061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.01778","last_updated":"2021-07-08T20:16:28Z","snapshot_observed_at":"2026-08-03T23:00:35.904734Z","submitted_at":"2021-04-05T05:26:29Z","title":"AST: Audio Spectrogram Transformer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.01778","snapshot_observed_at":"2026-08-07T01:05:44.293487Z","title":"AST: Audio spectrogram Transformer.arXiv preprint arXiv:2104.01778,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.293487Z"},"links":{"cited_paper":"/paper/2104.01778","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:6830f64a9e1f112ffdaf55f02140129325682233fd0ce9b91f75a357e475d28d","observation_id":"61012109-f531-4814-87f9-741cb0120e8d","resolution":{"observed_at":"2026-08-07T01:05:44.293487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.16691","last_updated":"2022-03-30T22:06:13Z","snapshot_observed_at":"2026-07-06T12:55:01.603047Z","submitted_at":"2022-03-30T22:06:13Z","title":"MAE-AST: Masked Autoencoding Audio Spectrogram Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.16691","snapshot_observed_at":"2026-08-07T01:05:44.036582Z","title":"MAE-AST: Masked autoencoding audio spectro- gram Transformer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.036582Z"},"links":{"cited_paper":"/paper/2203.16691","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:a03343d34a835a723847fcb7d19375772ac3cf709018fb6b437d21804a6e0d92","observation_id":"94486ff4-66e9-43fc-8068-e98be5dde473","resolution":{"observed_at":"2026-08-07T01:05:44.036582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.06695","last_updated":"2021-04-21T01:06:44Z","snapshot_observed_at":"2026-07-06T10:48:59.151429Z","submitted_at":"2021-03-11T14:32:33Z","title":"BYOL for Audio: Self-Supervised Learning for General-Purpose Audio Representation","version":2},"cited_work":{"arxiv_id":"2103.06695","doi":null,"metadata_source":"pith","pith_arxiv_id":"2103.06695","snapshot_observed_at":"2026-08-07T01:05:46.100971Z","title":"BYOL for Audio: Self-Supervised Learning for General-Purpose Audio Representation","venue":"eess.AS","work_id":"738e9f63-861a-444a-9b01-7db63a8d2cb4","year":2021},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.628445Z"},"links":{"cited_paper":"/paper/2103.06695","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:2e987ebfd766092880c2cce86da9d2398dd39db05f6df6a2f75e5f6d798c12ea","observation_id":"794bdb0a-bd06-4ee3-9d74-ceb86def731c","resolution":{"observed_at":"2026-08-07T01:05:46.163420Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.03497","last_updated":"2024-01-07T14:31:27Z","snapshot_observed_at":"2026-08-06T11:45:23.129204Z","submitted_at":"2024-01-07T14:31:27Z","title":"EAT: Self-Supervised Pre-Training with Efficient Audio Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.03497","snapshot_observed_at":"2026-08-07T01:05:44.124592Z","title":"Eat: Self-supervised pre- training with efficient audio transformer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.124592Z"},"links":{"cited_paper":"/paper/2401.03497","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:e57c72a062370d327f0ca889c042bd9dc216f68cc6da1124da8c706bca34c48b","observation_id":"0cb4f1d3-b331-46e7-8bf8-69e15104e12f","resolution":{"observed_at":"2026-08-07T01:05:44.124592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-07T01:05:44.165961Z","title":"An im- age is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.165961Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:9fe5563ff8e4a60d768484c0516ffb00725eb8ac7f6c91cf7061bf00dc178c25","observation_id":"8712af36-8e93-42c1-9bc3-40c4f46c2d24","resolution":{"observed_at":"2026-08-07T01:05:44.165961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15830","last_updated":"2024-01-11T13:16:43Z","snapshot_observed_at":"2026-07-06T16:53:05.949503Z","submitted_at":"2023-11-27T13:53:53Z","title":"A-JEPA: Joint-Embedding Predictive Architecture Can Listen","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.15830","snapshot_observed_at":"2026-08-07T01:05:44.234503Z","title":"11 Published as a conference paper at ICLR 2025 Jort F","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:44.234503Z"},"links":{"cited_paper":"/paper/2311.15830","citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:46661cb5497a12ad01725543281e200c67139df493b8a80e685d7521c5846312","observation_id":"4e0cbe5f-1dd4-4ca3-97cd-d682b4c8e7f0","resolution":{"observed_at":"2026-08-07T01:05:44.234503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:05:43.994900Z","title":"URL http://dx.doi.org/10.1109/TASLP","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","version":1},"reference_index":9304,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:43.994900Z"},"links":{"citing_paper":"/paper/2506.12222"},"observation_digest":"sha256:a726b487b30bf73a1ff68c78e3307eeb8c0950c98796cc7cf3b7226b436accf6","observation_id":"5031b894-4e5b-4ec3-bdfe-42f319320ced","resolution":{"observed_at":"2026-08-07T01:05:43.994900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.12222","last_updated":"2025-06-13T20:48:46Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-07T05:03:50.906639Z","submitted_at":"2025-06-13T20:48:46Z","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes"},"reference_resolution":{"displayed":26,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":21,"verified_exact":0,"verified_fuzzy":4},"total_outbound_references":26},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 26 of 26 outbound references and 2 inbound Pith citation observations for arXiv:2506.12222."}