{"as_of":"2026-08-06T03:30:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:24e50a274344430384e15dd5042bd6ef50abd33e60d4760b2d803c7105dd4455","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":5,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":5,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T08:50:37.709583Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T15:47:23.273387Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-07-13T17:37:12.659819Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2603.26292","last_updated":"2026-06-15T19:50:12Z","snapshot_observed_at":"2026-08-03T04:47:13.034793Z","submitted_at":"2026-03-27T11:03:08Z","title":"findsylls: A Language-Agnostic Toolkit for Syllable-Level Speech Tokenization and Embedding","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-13T17:37:12.659819Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2603.26292"},"observation_digest":"sha256:2b716a1dd3184222f3c9eb950ca35249342f9598da7f54b4088ab0cf14ee29ee","observation_id":"9107ecd9-b34d-4846-b3be-055b6f96949a","resolution":{"observed_at":"2026-07-13T17:37:12.659819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":"1904.03670","doi":null,"metadata_source":"pith","pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-07-10T15:47:23.273387Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","venue":"eess.AS","work_id":"4a03a99b-cf89-4a7e-8861-8e2917fd8ec2","year":2019},"citing_paper":{"arxiv_id":"2606.26556","last_updated":"2026-06-25T03:07:34Z","snapshot_observed_at":"2026-07-07T00:00:51.779266Z","submitted_at":"2026-06-25T03:07:34Z","title":"WQ-Fusion: Dynamic Gated Attention for Cross-Domain Audio Representation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-26T03:21:39.842959Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2606.26556"},"observation_digest":"sha256:e8d570aa61143aaa3bf9589494499f1124c6577fe164a331fac7921d1ec7d6fd","observation_id":"b66e635a-b0a0-4956-a1ae-7a1e5a142bb2","resolution":{"observed_at":"2026-07-04T14:39:57.212554Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":"1904.03670","doi":null,"metadata_source":"pith","pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-07-10T15:47:23.273387Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","venue":"eess.AS","work_id":"4a03a99b-cf89-4a7e-8861-8e2917fd8ec2","year":2019},"citing_paper":{"arxiv_id":"2607.07907","last_updated":"2026-07-08T20:42:46Z","snapshot_observed_at":"2026-08-02T21:44:55.258282Z","submitted_at":"2026-07-08T20:42:46Z","title":"Multimodal Unlearning Across Vision, Language, Video, and Audio: Survey of Methods, Datasets, and Benchmarks","version":1},"reference_index":249,"source":"arxiv_source","source_observed_at":"2026-07-10T15:38:58.361411Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2607.07907"},"observation_digest":"sha256:6e36556cafd9d23c4a9774e3c7d8370edfdba2cac3d57f01db0f14271b0e0ba3","observation_id":"741fbe33-76d3-4287-8649-b56b6537444b","resolution":{"observed_at":"2026-07-10T15:47:23.274542Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-08-01T01:19:07.656877Z","title":"Speech model pre-training for end-to-end spoken language understanding,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.25870","last_updated":"2026-07-28T15:37:44Z","snapshot_observed_at":"2026-08-06T01:47:39.152243Z","submitted_at":"2026-07-28T15:37:44Z","title":"VAD to the Bone: Ultra-Tiny Speech Activity Detection for Edge Deployment","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-01T01:19:07.656877Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2607.25870"},"observation_digest":"sha256:fab967121d5fa3eb74f289ab0464f14acc6512dceed66d06c2fc672a008aa150","observation_id":"c69cc7ce-597c-4dab-8348-63faafe4144f","resolution":{"observed_at":"2026-08-01T01:19:07.656877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-08-03T08:50:37.709583Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2607.29353","last_updated":"2026-07-31T12:37:48Z","snapshot_observed_at":"2026-08-05T23:13:16.167684Z","submitted_at":"2026-07-31T12:37:48Z","title":"Versatile On-device Adaptation at the Edge by Unifying Few-shot, Zero-shot, Continual, and In-context Learning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-03T08:50:37.709583Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2607.29353"},"observation_digest":"sha256:ef945048542cda015312fcd366c1e629fb6f5751fc0d80eab7dbe97ca1c1250a","observation_id":"f248c76e-2494-4794-a3f7-03a9e0cb401f","resolution":{"observed_at":"2026-08-03T08:50:37.709583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/1904.03670/citation-record","integrity":"/paper/1904.03670/integrity","json":"/paper/1904.03670/citation-record.json","paper":"/paper/1904.03670"},"outbound":[],"paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","latest_version":2,"primary_category":"eess.AS","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 5 inbound Pith citation observations for arXiv:1904.03670."}