{"as_of":"2026-08-08T01:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:db7a02cbcd0e9998b99045ae93c9a55ce9934e9cccb54e7a7a2959e3e7d25b32","coverage":[{"denominator":31,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:48:15.208113Z","state":"measured"},{"denominator":32,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":32,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:48:12.576291Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T12:48:15.322444Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"cited_work":{"arxiv_id":"2505.23509","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.23509","snapshot_observed_at":"2026-08-07T12:48:15.322444Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","venue":"cs.SD","work_id":"fc93cbf6-709a-4648-805c-604f43f1e3da","year":2025},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:12.576291Z"},"links":{"cited_paper":"/paper/2505.23509","citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:e48ba38b6a5d2596f5f1e784ec71fe2334f890844114f867e68c65b236872744","observation_id":"0aadf0c6-816f-41b8-8f4b-384280ddac7f","resolution":{"observed_at":"2026-08-07T12:48:15.415069Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.23509/citation-record","integrity":"/paper/2505.23509/integrity","json":"/paper/2505.23509/citation-record.json","paper":"/paper/2505.23509"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:21.477432Z","title":"A critical step in developing powerful audio deep neu- ral networks (DNNs) is converting audio signals into meaning- ful acoustic feature representations","venue":null,"work_id":"e1bc2823-5bc2-4cae-9bc6-9dc636f73b14","year":null},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:12.527319Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:39709f283d59a84af7ce571f8355b7a2cf0b5f3c6568e46aedec5619d5bf266f","observation_id":"8544dcc9-e87f-4db7-9eb8-e673504d47a4","resolution":{"observed_at":"2026-08-07T12:48:21.659608Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"cited_work":{"arxiv_id":"2505.23509","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.23509","snapshot_observed_at":"2026-08-07T12:48:15.322444Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","venue":"cs.SD","work_id":"fc93cbf6-709a-4648-805c-604f43f1e3da","year":2025},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:12.576291Z"},"links":{"cited_paper":"/paper/2505.23509","citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:e48ba38b6a5d2596f5f1e784ec71fe2334f890844114f867e68c65b236872744","observation_id":"0aadf0c6-816f-41b8-8f4b-384280ddac7f","resolution":{"observed_at":"2026-08-07T12:48:15.415069Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:21.178649Z","title":null,"venue":null,"work_id":"48c63652-ceae-45cc-8c39-720f8fc525fb","year":null},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:12.637154Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:5011aecfa3a6d2e0b87dee596723460463dfc3fdef30028c3e5c6e347135abda","observation_id":"f2750332-72bc-4a06-bf00-a49fa2093d16","resolution":{"observed_at":"2026-08-07T12:48:21.355547Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:20.942479Z","title":null,"venue":null,"work_id":"7ddb9123-45e0-4286-8497-1f7e2e6d0e54","year":null},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:12.686942Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:f826453c9fb1812ec13091f15c5992bf0c322df425e46152296260ae80d0a814","observation_id":"24d82cb6-04c9-4e60-9ef4-d5c3262ff414","resolution":{"observed_at":"2026-08-07T12:48:21.029035Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:20.722809Z","title":null,"venue":null,"work_id":"740d678d-b206-4486-b0c7-5ca0d207ac62","year":null},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:12.795271Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:086cc74a68980c457d24c9627ce7be98807642cc39d28c8d1320254f84b3d85e","observation_id":"8675b5d3-a907-408a-acc6-b2e0b0a43027","resolution":{"observed_at":"2026-08-07T12:48:20.782018Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:20.539359Z","title":"A survey of audio classification using deep learning,","venue":null,"work_id":"e0699601-ba1c-4d0a-ac30-9c31fdba480f","year":2023},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:12.887256Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:7faa3f9ae929c9b4c83f086d0c589629d07c6ea278f3924c8ce1d1f696328908","observation_id":"fe22b275-bcc3-4752-a3a7-eac1d6bcf3e4","resolution":{"observed_at":"2026-08-07T12:48:20.618642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:20.374858Z","title":"AST: Audio spectrogram transformer,","venue":null,"work_id":"5553c8e0-6eae-400d-b585-f966140a5ab2","year":2021},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:13.000445Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:a56fd8ae244b002f9ecfe5ddff02f22a8b0ac07c07ce1a6060369df894ea2ef3","observation_id":"3c3ad048-3e4c-406b-9f9f-526af2a1075f","resolution":{"observed_at":"2026-08-07T12:48:20.457969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:20.196110Z","title":"Audio set: An ontology and human-labeled dataset for audio events,","venue":null,"work_id":"901141e7-5e2f-4ef8-a439-bb6762b899a2","year":2017},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:13.108658Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:8d9259992e2ed1b6344e4d21ba665a4500118924b9af887906f639c934c7e5e4","observation_id":"4d6b88ac-86f1-4ff6-b5ae-eed6d1bcaed5","resolution":{"observed_at":"2026-08-07T12:48:20.298087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:20.042115Z","title":"CNN architectures for large-scale audio classification,","venue":null,"work_id":"dc99a74d-ac85-427d-b840-f29384c1d1c2","year":2017},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:13.169301Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:ba429a66b1982a57575679135d6867b2b75df1b9fdd3a4cadc7a8ae045346aca","observation_id":"1a8f05df-e420-4517-8438-6cb96e23429b","resolution":{"observed_at":"2026-08-07T12:48:20.108462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:19.871288Z","title":"Richard, V","venue":null,"work_id":"8c42a035-2cb1-4bb4-80a4-33e16fa721ef","year":2025},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:13.264884Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:ad5b36d56254e151cc0446ca1ca02a4a97e633199cbefb6091967baaf0165625","observation_id":"5e6ade39-ae94-47c4-a668-e71ca072b41d","resolution":{"observed_at":"2026-08-07T12:48:19.940343Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:19.677935Z","title":"Robust DOA esti- mation from deep acoustic imaging,","venue":null,"work_id":"f58fa2d7-a792-4db9-90ac-01c97ade9691","year":2024},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:13.337821Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:aac6df63e41c8bad84111647de9da3b7d52fb3790059a0c85698c034d43cf207","observation_id":"40e6dc28-1cd9-4d32-9a18-9f5a3ab4abbb","resolution":{"observed_at":"2026-08-07T12:48:19.766823Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:19.441018Z","title":"A review of differentiable digital signal processing for music and speech synthesis,","venue":null,"work_id":"ad8e2ff1-c034-4ac7-ac7b-f973f49eddbb","year":2024},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:13.463863Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:5e70ff17d2eb327dbf6558a0e770d181b62cfcb589b3eabefc447fa7051da7d4","observation_id":"ea9ece3b-248c-422b-b87c-8eede74c9d04","resolution":{"observed_at":"2026-08-07T12:48:19.583950Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:19.047814Z","title":"The modulation spectrogram: In pursuit of an invariant representation of speech,","venue":null,"work_id":"aa366994-d923-4d15-b9d9-bb170acd4ae6","year":1997},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:13.557074Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:cee256c4db300ed73a111d924a4b82c240840a5b587ad13989ce9217d2d9e00e","observation_id":"7f2f58c9-283b-4aca-960a-a70352002351","resolution":{"observed_at":"2026-08-07T12:48:19.201080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:18.806556Z","title":"Speech intel- ligibility prediction using spectro-temporal modulation analysis,","venue":null,"work_id":"29231cd9-2863-4f26-b732-71ffae010726","year":2020},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:13.640105Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:e518ab48f89cbf8969e85618b7508868df68b6c19aeff85f4682bfa2230f851c","observation_id":"877c4e90-4e92-45e6-b052-e052aa694581","resolution":{"observed_at":"2026-08-07T12:48:18.957448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:18.602519Z","title":"Com- paring different flavors of spectro-temporal features for ASR","venue":null,"work_id":"2fdf5f13-5f7a-4231-90ac-5c8953035fa2","year":2011},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:13.748977Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:df492172d912ab22d9d7c034a52daf969b0766f6bd9997d14c724ad93a8a1341","observation_id":"dcebb7a8-0ab8-4381-9f6a-33e0796c1f89","resolution":{"observed_at":"2026-08-07T12:48:18.683905Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:18.375541Z","title":"Automatic mu- sic genre classification based on modulation spectral analysis of spectral and cepstral features,","venue":null,"work_id":"13518897-24db-416d-9790-987ab56431c7","year":2009},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:13.856716Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:7e6bd626753d09e6f4fbc66b782e3f066fe7578fb26e7db9b2ddfe5ebc5d4229","observation_id":"8e4bf80a-5f0b-4b2b-9399-418f04723c79","resolution":{"observed_at":"2026-08-07T12:48:18.486388Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:18.190788Z","title":"Discrimination of speech from nonspeech based on multiscale spectro-temporal modulations,","venue":null,"work_id":"e587f6dd-ee0f-4156-abfa-4a176d2af6b5","year":2006},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:13.932886Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:2f69c79ecc43e35723166caa50441dc6e8c02581881d7bd3d8b59b5ed7dd662d","observation_id":"356e24d9-5f44-49f9-8696-51733ebf700c","resolution":{"observed_at":"2026-08-07T12:48:18.289491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:18.030426Z","title":"Distinct sensitivity to spectrotemporal modulation supports brain asym- metry for speech and melody,","venue":null,"work_id":"54fa9980-dd1d-4941-9b4b-55617db8bc9d","year":2020},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:14.016047Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:31cbecaa2d37e12bf92e15105fe0ee62d4ec33c7e08fcc792219c4417ec0ec4e","observation_id":"69ea92cb-6151-4cd2-8565-9a4b2a858655","resolution":{"observed_at":"2026-08-07T12:48:18.124490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:17.693041Z","title":"Spectrotemporal modulation provides a unifying framework for auditory cortical asymmetries,","venue":null,"work_id":"9d8b3735-e485-45ac-8401-7a215533df7a","year":2019},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:14.110632Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:251e24dd32e222b042d3cc361278d349ce573bb48794211a5917f4fe55b2ad9c","observation_id":"b5adde92-8f18-4045-ad0d-a933daca95a4","resolution":{"observed_at":"2026-08-07T12:48:17.831800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:17.526875Z","title":"Spectro-temporal acoustical markers differenti- ate speech from song across cultures,","venue":null,"work_id":"bb31ea71-f153-42e9-b742-30b98674d22a","year":2024},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:14.219594Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:aec9ff25fda851285d4d3fc8165697bfadc5d3894e558dc7cac8ec04a24e7cbe","observation_id":"835fff49-c290-46c4-a377-b5dc8068f740","resolution":{"observed_at":"2026-08-07T12:48:17.576315Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:17.386448Z","title":"The human auditory system uses amplitude modulation to distinguish music from speech,","venue":null,"work_id":"18f9b938-490b-4c37-a993-28c57a8b2c69","year":2024},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:14.287739Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:49abf63553e6aecc30c46543bf41e17c1683c87ee5947d838189f5f40f4c7d11","observation_id":"21b5f124-a30b-4780-a68e-532498bba7ad","resolution":{"observed_at":"2026-08-07T12:48:17.472229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:17.194892Z","title":"Distinct cortical pathways for music and speech revealed by hypothesis-free voxel decomposition,","venue":null,"work_id":"e2292235-d8c2-4903-a4e3-6b08a66943b1","year":2015},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:14.411085Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:7e57f82ef432a7e6240bb5050c1428758942a7f5605dc7adad42f5904a6e838e","observation_id":"4242752c-6a66-456a-a2c9-c71f7ea44354","resolution":{"observed_at":"2026-08-07T12:48:17.286093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:16.993699Z","title":"Hybrid Transformers for Music Source Separation,","venue":null,"work_id":"efa1ff87-baf0-4d37-b876-cf71588d92fd","year":2023},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:14.487770Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:64bee64ee17509bf05cb1fbe302beffc7d9d0abeeb440ec16d059a5a9ddefcb9","observation_id":"90697415-7289-4fb8-8c05-010c665b4efd","resolution":{"observed_at":"2026-08-07T12:48:17.082844Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:16.801304Z","title":"SONYC Urban Sound Tagging (SONYC-UST): a multilabel dataset from an urban acoustic sen- sor network,","venue":null,"work_id":"9963d1c9-bb04-4c93-a5a9-a0d04817350e","year":2020},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:14.578272Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:df9793c5100a14da9098f474444dad54305da427983c76fffbff257d776618bc","observation_id":"76195d72-4c58-4a45-9e20-519b6659d383","resolution":{"observed_at":"2026-08-07T12:48:16.930375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:16.568872Z","title":"eBird: A citizen-based bird observation network in the biological sciences,","venue":null,"work_id":"bf62c9d3-8349-40ae-8305-b38c401ba27c","year":2009},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:14.692285Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:38be3b276a1035a8b29de90dc01e9dfc74cb98655d2f6e840c1c5484c0b523f4","observation_id":"2a17fc99-58f8-456f-831c-f80d8b0c6432","resolution":{"observed_at":"2026-08-07T12:48:16.682786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:16.392400Z","title":"The cortical organization of speech processing,","venue":null,"work_id":"8fd606a1-f0af-4906-bde7-d92c030bb752","year":2007},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:14.762739Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:dd56b4af1b168f1dfd64f8c249d63cb5918efd7a468ba23b18292e275bd965a0","observation_id":"1245d9cd-6b81-4dc6-87fa-a517e5e32531","resolution":{"observed_at":"2026-08-07T12:48:16.463533Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:16.195551Z","title":"Facing Imbalanced Data Recommendations for the Use of Performance Metrics,","venue":null,"work_id":"4f03a1ad-7c13-4104-b8d9-c01295c5d180","year":2013},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:14.867868Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:58ffe21b18370cd975bd53c5a22e556acf4f5d5d1aa7dd5108edd73581a31fb2","observation_id":"0f0ad36d-613d-47db-8681-092911200ac5","resolution":{"observed_at":"2026-08-07T12:48:16.270418Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:16.032842Z","title":"Survey on deep learning with class imbalance,","venue":null,"work_id":"f25ebf2b-0f4e-48bb-8d6a-3cd50d67f8b6","year":2019},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:14.936404Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:842e55611629eb1be2ea8722b2ad89010593e15f657c320f93bf52bd6c36e2ba","observation_id":"86b0af17-e0d9-46a4-a594-1bca5ffc8f7c","resolution":{"observed_at":"2026-08-07T12:48:16.114290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:15.845470Z","title":"Many but not all deep neural network audio models capture brain responses and exhibit correspondence between model stages and brain regions,","venue":null,"work_id":"0e1312eb-fe2e-457e-a087-6c6c1ef2948c","year":2023},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:15.039936Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:62bffba42f1aca80c35c51f70d511a6d45a5cd73466a08babc05c62ec686f292","observation_id":"9801d646-ddce-47f4-9124-7696b9d7ee52","resolution":{"observed_at":"2026-08-07T12:48:15.922809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:15.661088Z","title":"Reconstructing the spectrotem- poral modulations of real-life sounds from fMRI response pat- terns,","venue":null,"work_id":"31115d07-3566-4502-b35c-790058917abc","year":2017},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:15.140447Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:b3604b8fbd3684118d65bd43197c757e0eadca0eab21e234ac3981e2daf255d2","observation_id":"10cc63e4-d7eb-4b2f-b089-2da056a4b2ed","resolution":{"observed_at":"2026-08-07T12:48:15.768965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:48:15.487377Z","title":"Decoding spectrotemporal features of overt and covert speech from the hu- man cortex,","venue":null,"work_id":"1043f71c-7a75-4e22-ad0b-da2437f93a91","year":2014},"citing_paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:15.208113Z"},"links":{"citing_paper":"/paper/2505.23509"},"observation_digest":"sha256:aff1c16d833593cbc6ebc688c310d5e498551a9bf8d935538f98804ad3c69013","observation_id":"5d5eec24-f548-4559-977b-dfe74a136534","resolution":{"observed_at":"2026-08-07T12:48:15.598584Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.23509","last_updated":"2025-05-29T14:52:47Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-07T12:42:22.760991Z","submitted_at":"2025-05-29T14:52:47Z","title":"Spectrotemporal Modulation: Efficient and Interpretable Feature Representation for Classifying Speech, Music, and Environmental Sounds"},"reference_resolution":{"displayed":31,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":3,"verified_exact":1,"verified_fuzzy":27},"total_outbound_references":31},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 31 of 31 outbound references and 1 inbound Pith citation observation for arXiv:2505.23509."}