{"as_of":"2026-08-16T23:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5f65545dd4a115487ea1ae73323e081ff7243ffd1970e2e1948d5a99fe2391b9","coverage":[{"denominator":53,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":53,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:03:18.031484Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:03:14.249592Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T20:03:18.073334Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"cited_work":{"arxiv_id":"2507.03912","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.03912","snapshot_observed_at":"2026-08-06T20:03:18.073334Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","venue":"eess.AS","work_id":"9a5aa0f7-f98c-41ca-aec8-7557d710dae2","year":2025},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:14.249592Z"},"links":{"cited_paper":"/paper/2507.03912","citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:21136e38ae0b96e7f29867254e3e8b9070155dcc8adea1d09311232c94d2aa0f","observation_id":"a8e9006e-c63d-4fb7-8a0a-0c41862ebe45","resolution":{"observed_at":"2026-08-06T20:03:18.076461Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.03912/citation-record","integrity":"/paper/2507.03912/integrity","json":"/paper/2507.03912/citation-record.json","paper":"/paper/2507.03912"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"cited_work":{"arxiv_id":"2507.03912","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.03912","snapshot_observed_at":"2026-08-06T20:03:18.073334Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","venue":"eess.AS","work_id":"9a5aa0f7-f98c-41ca-aec8-7557d710dae2","year":2025},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:14.249592Z"},"links":{"cited_paper":"/paper/2507.03912","citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:21136e38ae0b96e7f29867254e3e8b9070155dcc8adea1d09311232c94d2aa0f","observation_id":"a8e9006e-c63d-4fb7-8a0a-0c41862ebe45","resolution":{"observed_at":"2026-08-06T20:03:18.076461Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.567080Z","title":"[8] have also conducted automatic prosody an- notation for constructing a speech synthesis database, similar to the objective of our study","venue":null,"work_id":"a7ef8801-ddbe-45ef-9714-3c56a805237b","year":null},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:14.334997Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:fec3616b1ec97886e8ce64366aa8437dd4c352134317a329521cca776c0eedba","observation_id":"7a37b1f9-63a0-4b0b-b6d1-b84466d39149","resolution":{"observed_at":"2026-08-06T20:03:18.570098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.559762Z","title":"ashita wa hare desuka","venue":null,"work_id":"b70a3408-55fd-4347-a05e-cc9214e63d92","year":null},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:14.448986Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:2797344442580d4a09c716b50f72b1a70fedc634887a821b4e5455eeea0a61e8","observation_id":"486d9ad2-4af9-4aea-bb4d-14ec30cd5b01","resolution":{"observed_at":"2026-08-06T20:03:18.562280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.551170Z","title":"Figure 3 shows the outline of the proposed model","venue":null,"work_id":"d3d025e9-0c37-40ad-ae61-c0361593378d","year":null},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:14.576338Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:04f67b4d8294a518753a761d38c4b439748ee5a1ea6719a2df5458a0ab71c523","observation_id":"25f7c816-b74d-4ccd-bd15-7b8514a2d4d3","resolution":{"observed_at":"2026-08-06T20:03:18.554087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.543569Z","title":"*”: Other symbol that was neither a high-low transition nor a phrase boundary. • “[","venue":null,"work_id":"2a2fced8-f35e-4b2a-a8b8-4fa61327830e","year":null},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:14.718140Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:09f76a12a6da128b79c27cd131fb718649a7387bfe6397f5e76794954f6fcc5d","observation_id":"7f5ba96d-e444-4a68-8274-f3d160d541e9","resolution":{"observed_at":"2026-08-06T20:03:18.546322Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.527824Z","title":null,"venue":null,"work_id":"a6c776cc-9505-4bd4-b498-fef34b6d242b","year":null},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:14.941364Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:4f4183eeed28b097692d2ac14d2d9016a4401ea3a43b19c8bc530939cbd273be","observation_id":"d85e9b1d-ba7e-45fe-98fd-6753788a252e","resolution":{"observed_at":"2026-08-06T20:03:18.530571Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.471612Z","title":"Automatic prosody annotation with pre-trained text- speech model,","venue":null,"work_id":"38cad07c-02c2-429d-abb7-54f47dd5c845","year":2022},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.460095Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:4a764dfe4cd8641f6fc26f0888244ea00f34f994b5c94a1c60d5fbdf6952b9f6","observation_id":"ad3aa484-253a-4abe-8ef5-5b8e17d39938","resolution":{"observed_at":"2026-08-06T20:03:18.474594Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.520162Z","title":"Mora-level prosody prediction for text-to-speech us- ing japanese BERT without accentual labels,","venue":null,"work_id":"46e33b26-56a9-4526-b29e-89dac69850b8","year":2025},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:14.995512Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:5814eef22c34a30e77f123d0791c5fa87af6291f1b415cb4e351dca8648036fa","observation_id":"24699a4f-1edb-44b9-aef4-f4cd2e9745ab","resolution":{"observed_at":"2026-08-06T20:03:18.522972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.512165Z","title":"Semi-supervised prosody model- ing using deep Gaussian process latent variable model,","venue":null,"work_id":"93ab933e-7104-4e53-8520-2fa746bd6f78","year":2019},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.059899Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:e96fd1ac7a94896125ac9f374c0730a245a8ea2fd19aae54d0822f79b2adeb56","observation_id":"91978ff0-5f1a-43d0-8fac-dce5a42e6dec","resolution":{"observed_at":"2026-08-06T20:03:18.515166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.503521Z","title":"Ac- cent modeling of low-resourced dialect in pitch accent language using variational autoencoder,","venue":null,"work_id":"ade80434-b53e-4931-a0ce-bab6ef26e190","year":2021},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.121236Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:fe441d4bce54f093c41cbaab4b0d1ca5f9becdf23fc5d08e0601e69aa5983e72","observation_id":"2800bd16-57a5-4d68-b532-08a393ce229a","resolution":{"observed_at":"2026-08-06T20:03:18.506457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.495657Z","title":"Cross-dialect text- to-speech in pitch-accent language incorporating multi-dialect phoneme-level BERT,","venue":null,"work_id":"45418b57-cd60-448a-97b0-5c3fffcd2586","year":2024},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.224190Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:b0bbc1e126331db1a83580780aeef2b6bc424a66dd42053a2f3a779f0f340864","observation_id":"17c90688-3bf3-4167-b89c-4e05771cc5d0","resolution":{"observed_at":"2026-08-06T20:03:18.498411Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.487753Z","title":"Multi-modal automatic prosody anno- tation with contrastive pretraining of speech-silence and word- punctuation,","venue":null,"work_id":"4f05e347-494b-462a-a781-a9e3797845ba","year":2024},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.277753Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:f0b952ff20fcbbabd032556d3a29c0300dd468da3fb4ab2a25cfb04a37c1ee6b","observation_id":"7752be24-91c1-4332-92de-fe68656049ef","resolution":{"observed_at":"2026-08-06T20:03:18.490401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.479905Z","title":"Language-independent prosody-enhanced speech representations for multilingual speech synthesis,","venue":null,"work_id":"7367e5c3-817e-42d0-9a55-bbae0b7d6163","year":2024},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.376794Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:6596b1758728cca24138dc3bc001a77a35619ad8cd2748cf2b659103bc526765","observation_id":"9f4f8240-b7e3-4f27-b812-aa1f6842bcc1","resolution":{"observed_at":"2026-08-06T20:03:18.482741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.416355Z","title":"BERT: Pre- training of deep bidirectional transformers for language under- standing,","venue":null,"work_id":"65da0005-fa53-4eb0-8034-d70bc411de51","year":2019},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.971177Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:a0d189ea950fe6323f344a76607bd1036a316f0d5a0178d9e447ec86379285a1","observation_id":"8377d84f-481f-43c6-b405-5126743c9a7e","resolution":{"observed_at":"2026-08-06T20:03:18.419024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.463090Z","title":"Audio- conditioned phonemic and prosodic annotation for building text- to-speech models from unlabeled speech data,","venue":null,"work_id":"019baa38-bcb4-4797-90f7-d23fe97a98c9","year":2024},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.540553Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:571a61858ca02e752bcc2a1a099302382c1bfc18aff0984c5d45479966e49dd1","observation_id":"501b9959-6d09-43de-bc6f-d6a5715126f7","resolution":{"observed_at":"2026-08-06T20:03:18.465831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.454882Z","title":"Low-resourced phonetic and prosodic feature estimation with self-supervised-learning-based acoustic modeling,","venue":null,"work_id":"62bddb35-e163-4786-880f-5de6992459fc","year":2024},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.611761Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:ecde0a2d386009deb2713bd24e527068e6ec16cfd0d61010ddef5564c220af25","observation_id":"02fafbbd-40ca-4ff9-8c6b-9d4d91b2a8c8","resolution":{"observed_at":"2026-08-06T20:03:18.457670Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.446023Z","title":"Cross-lingual speech- based ToBI label generation using bidirectional LSTM,","venue":null,"work_id":"fb61cdd3-ba98-4d65-876f-3dd43d594c57","year":2019},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.676428Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:948fc38cf6d2c9ac6bfe2c3aa75b54b991b1f6bb80d797be9082821b56c93063","observation_id":"3a4335c3-5762-4f48-90b7-61b6610ca6c1","resolution":{"observed_at":"2026-08-06T20:03:18.449384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.437853Z","title":"A unified accent esti- mation method based on multi-task learning for Japanese text-to- speech,","venue":null,"work_id":"c527025a-90bb-430a-ba69-349af7d815a8","year":2022},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.742927Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:8734ca2e75ff79e0ec7f977f27277e431253a15677324501b07ad48abe0783a6","observation_id":"994affb5-0eb8-48ed-8edd-474a1541b990","resolution":{"observed_at":"2026-08-06T20:03:18.440586Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.429805Z","title":"Wav2ToBI: a new approach to automatic ToBI transcription,","venue":null,"work_id":"d47588da-a0df-4549-aeb9-1fab232e9158","year":2023},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.809588Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:ee4af513ef1242d53a5041f6bd8ec76260bf45420f82cab5c249bf1a2ea73866","observation_id":"73a9539d-6bdd-4c62-bbfc-1831c6547496","resolution":{"observed_at":"2026-08-06T20:03:18.432635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:15.886329Z","title":"Conformer: Convolution-augmented transformer for speech recognition,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:15.886329Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:fa3fbe3992146cfbc67ed671927964afb48fb44f3f2ef29c11aca7eb101a502b","observation_id":"7d985065-12a4-469d-ae23-b642b0469279","resolution":{"observed_at":"2026-08-06T20:03:15.886329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.358682Z","title":"Investigation of enhanced Tacotron text-to-speech synthesis systems with self- attention for pitch accent language,","venue":null,"work_id":"4abbdf39-f767-447e-9290-e50e286736c3","year":2019},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.479507Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:39b5a86fcb8daf26a28b29e49c10b770dc929871ae7bd12063e2203ef2596865","observation_id":"75ce9da2-3cf3-4a12-8a6a-6f6eaa5606a5","resolution":{"observed_at":"2026-08-06T20:03:18.361910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.408670Z","title":"Robust speech recognition via large-scale weak su- pervision,","venue":null,"work_id":"8b39ca30-a3c5-4dc7-9b0c-dad22537a5a7","year":2023},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.049594Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:400fb3853e028092354c86d66b5547ffd0dcea140968701a7507d3d99fc40f2b","observation_id":"5b927204-f5ad-474e-aaba-5cad605fe1e3","resolution":{"observed_at":"2026-08-06T20:03:18.411280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.400519Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech representa- tions,","venue":null,"work_id":"fbd4b44b-6472-4cfe-a47f-742bbd2468be","year":2020},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.124677Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:cd61c5127c1afe8608928e7a242ac19aec589a5d5dbddf3469c7e19f57614f9b","observation_id":"01d1492d-4e95-403e-88d1-d12a7e03f4e1","resolution":{"observed_at":"2026-08-06T20:03:18.403453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.392695Z","title":"Pre-trained text embeddings for enhanced text-to- speech synthesis,","venue":null,"work_id":"2c555ed8-f2a4-4e7d-8bc8-ae485f7e4466","year":2019},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.187592Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:dd9d6c174ebf5fdbeaebcdb2a5dd1250e1c5a8492fc9133771735bbd4c8a2e98","observation_id":"e28cb59e-5e18-41cf-9318-cede3a3f7cf9","resolution":{"observed_at":"2026-08-06T20:03:18.395326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.384640Z","title":"Improving prosody with linguistic and BERT derived features in multi-speaker based Mandarin Chinese neural TTS,","venue":null,"work_id":"4f97c072-0c3a-4aa3-a79f-a9a8a72bb7d8","year":2020},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.256372Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:889a1646b467d140b445947e15adeecb233b9668a380e6197922fbb75adee674","observation_id":"91b6ee43-c70f-4403-9818-071a4513fae6","resolution":{"observed_at":"2026-08-06T20:03:18.387703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.376614Z","title":"Improving the prosody of RNN-based English text-to-speech synthesis by incorporating a BERT model,","venue":null,"work_id":"342074d0-0588-4a5e-9b6c-7e011d21ba14","year":2020},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.342536Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:ba59e0568f4ad43e8252c54b293480373d235ef79b781e77ec604a72c9ecfac0","observation_id":"0596f701-8685-4c3f-93ef-90ef31690aa2","resolution":{"observed_at":"2026-08-06T20:03:18.379448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.367690Z","title":"Im- proving prosody modelling with cross-utterance BERT embed- dings for end-to-end speech synthesis,","venue":null,"work_id":"2e119217-bc99-424b-9070-32de14f303e1","year":2021},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.410019Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:9a294b3c8763680aa9e91f7e8991eb1b77f0ea9b8a60ed52d0df1fe2169eb0e7","observation_id":"e1b73488-3522-47eb-b05f-08f9e55a9403","resolution":{"observed_at":"2026-08-06T20:03:18.371358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.312723Z","title":"WavLM: Large-scale self-supervised pre-training for full stack speech processing,","venue":null,"work_id":"809b7075-91ed-4bb5-ad26-a565629d69c0","year":2022},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.032913Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:d5bf844e49641e2e101626d6909c0634aafd9f7105ed33620e9d1e5ff9d93c2a","observation_id":"cc6a775b-2392-47c7-b5e8-6c6e12926128","resolution":{"observed_at":"2026-08-06T20:03:18.315366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.350192Z","title":"PE- Wav2vec: A prosody-enhanced speech model for self-supervised prosody learning in TTS,","venue":null,"work_id":"b287c3dd-50ba-4ab3-849c-b7e015ccc75b","year":2024},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.565219Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:01b9ee3f6c63175292ac2fde5e00d85ba87225aa51cf775259682b8512e0218a","observation_id":"57503311-5675-44f1-bb7e-bd5a1c06b197","resolution":{"observed_at":"2026-08-06T20:03:18.353407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.342094Z","title":"PnG BERT: Aug- mented BERT on phonemes and graphemes for neural TTS,","venue":null,"work_id":"48c4d20d-9500-46e4-b876-d2ff34c76e2f","year":2021},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.622279Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:56978749bba35f3adc933ccf9e4cd6684f60384b5ee75e1cbffa4180d22d7c8a","observation_id":"f97fc26c-1ad4-4f1c-8e71-e2eda8f71880","resolution":{"observed_at":"2026-08-06T20:03:18.344923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.08810","last_updated":"2023-01-20T21:36:16Z","snapshot_observed_at":"2026-08-16T16:01:03.727685Z","submitted_at":"2023-01-20T21:36:16Z","title":"Phoneme-Level BERT for Enhanced Prosody of Text-to-Speech with Grapheme Predictions","version":1},"cited_work":{"arxiv_id":"2301.08810","doi":null,"metadata_source":"pith","pith_arxiv_id":"2301.08810","snapshot_observed_at":"2026-08-06T20:03:18.058783Z","title":"Phoneme-Level BERT for Enhanced Prosody of Text-to-Speech with Grapheme Predictions","venue":"cs.CL","work_id":"9ab6671e-4fd2-4b91-b7df-57a928e4362e","year":2023},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.688013Z"},"links":{"cited_paper":"/paper/2301.08810","citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:eec27d8c6ced99058885df929a9bedafb4f53b7f58ada2aa2414f895d2beda8d","observation_id":"b8b7e871-807a-4ee9-9e2b-a67c10d1f833","resolution":{"observed_at":"2026-08-06T20:03:18.063898Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.334182Z","title":"Detection of prosodic bound- aries in speech using wav2vec 2.0,","venue":null,"work_id":"eba8f99c-f866-4100-9cdd-af00ed8d61db","year":2022},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.765347Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:4b245ccc926b535c760eb93a4d604ca627f443f07259b7d7e6e0af9b5fc61a8c","observation_id":"acfc20ea-a1a6-4a01-8151-3d210e10a9e6","resolution":{"observed_at":"2026-08-06T20:03:18.336829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.326194Z","title":"Corpus of Spontaneous Japanese: Its design and evaluation,","venue":null,"work_id":"34a9aa2f-3e88-400b-8920-c45fe9151996","year":2003},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.852515Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:e2dbd85dbdd961a3de6cb8689341cdf29c97df45c6addb4d389be07ea0b7531c","observation_id":"0f49b92c-8d03-47be-89cf-189aa850975d","resolution":{"observed_at":"2026-08-06T20:03:18.328914Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:16.935390Z","title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:16.935390Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:92703017ecf84ebf1dc341d9d63da42671b3a08d46b662812ce201ec1cfce345","observation_id":"060b2b47-c026-4190-bffb-129380289285","resolution":{"observed_at":"2026-08-06T20:03:16.935390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.258597Z","title":"V AE-based phoneme alignment using gradient an- nealing and SSL acoustic features,","venue":null,"work_id":"9afe94ef-111e-4c88-9e7e-d0c034da11ec","year":2024},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.608199Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:39615e3a4a3900b3c4c42234d2f0e2a7f2809877c387cbe6adf89ca20f47370e","observation_id":"a4060926-a9e1-4dc4-9ad1-c32572eba2a7","resolution":{"observed_at":"2026-08-06T20:03:18.261266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.305186Z","title":"Self-supervised speech representation learning: A review,","venue":null,"work_id":"99154df1-b9e9-4be5-985c-9cf5ecbae5db","year":2022},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.101847Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:b218a88df2c761ce2b02c5f971208ba142eb50ed43ce4c72640214eb2ebfcc24","observation_id":"c7b819f2-d590-4756-a8f6-d33de3c60e10","resolution":{"observed_at":"2026-08-06T20:03:18.307988Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.297418Z","title":"StyleTTS 2: Towards human-level text-to-speech through style diffusion and adversarial training with large speech language models,","venue":null,"work_id":"69d95745-d7e7-4ed0-a9e4-69e5d9d49ffa","year":2023},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.116416Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:21b996c20b27154ba9fdd780747b646bb2d07ded50d8e65379e90598084e0353","observation_id":"abc9ed54-b6e4-446c-8c81-fef129ee3414","resolution":{"observed_at":"2026-08-06T20:03:18.300322Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.289367Z","title":"Emobox: Multilingual multi-corpus speech emotion recognition toolkit and benchmark,","venue":null,"work_id":"a70bcdc1-0533-48e7-a2dc-4583ea9a2355","year":2024},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.161090Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:be49674ed222b5d88e571dac558934f6c908f482c6d90ba01caf566cf6b037fe","observation_id":"19f23418-8170-4b75-a43d-390fc79a55a4","resolution":{"observed_at":"2026-08-06T20:03:18.292565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.281700Z","title":"Non-intrusive speech intelligibility pre- diction for hearing-impaired users using intermediate ASR fea- tures and human memory models,","venue":null,"work_id":"81be61a1-e378-46c1-9161-315eab3cb267","year":2024},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.249134Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:0c687077904799390e00f9cbc44f52fad8e797f4b0caad8e425f30a600e9ee3d","observation_id":"2e81d025-b644-4923-bc15-00202f2e2917","resolution":{"observed_at":"2026-08-06T20:03:18.284318Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.273996Z","title":"Montreal forced aligner: Trainable text-speech align- ment using kaldi","venue":null,"work_id":"92ff51d7-b700-406b-aef1-cb0fb2c68034","year":2017},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.361914Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:7f01bb1f4a703d5a04cecf94fa2a0bbcdcd614dafd3e07ceddaedd5b3643497b","observation_id":"0b479e7d-188f-4172-bba5-2fae76a05cfe","resolution":{"observed_at":"2026-08-06T20:03:18.276771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.266596Z","title":"One TTS alignment to rule them all,","venue":null,"work_id":"50dc3105-db9d-4837-8799-9bb248058457","year":2022},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.465887Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:bac71cc9d88bea34cdc30162180220ee7e0c17f425389dd390bdd2b293602ab2","observation_id":"36e6563f-7f5b-4834-8ede-aff580a51a1c","resolution":{"observed_at":"2026-08-06T20:03:18.269135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.205155Z","title":"X-JToBI: an extended J-ToBI for spontaneous speech,","venue":null,"work_id":"5e05dde7-360f-45f1-8a33-60ce744ea385","year":2002},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:18.020758Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:206a44825878f2af7d62651806cad8ad7c61c502613bb0a6c9312d9efb17582c","observation_id":"6fe82536-100d-48da-b81a-b35f370b1b9a","resolution":{"observed_at":"2026-08-06T20:03:18.207838Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.251256Z","title":"Stress in Thai,","venue":null,"work_id":"78e97a6c-7010-4d6a-a4ee-a32d7e784094","year":1986},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.750478Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:987be79e0beafa8134f00048ac2a4a4b107273290aa67bdcd4ac14072a934108","observation_id":"1e7d9279-efec-43b0-a8ed-86e58662a4a5","resolution":{"observed_at":"2026-08-06T20:03:18.253843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.535576Z","title":"ee shumi ga ongaku nandesu keredomo","venue":null,"work_id":"cc442a71-992f-46bf-8ac6-28e82769ad89","year":null},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:14.805676Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:dee1b6c5a1afb6fca67af3ca9dfddbbf49d44565d33ae1bd6b9d3ee3e335b7ee","observation_id":"30199321-7b5d-41a0-9bdd-cedac8f77223","resolution":{"observed_at":"2026-08-06T20:03:18.538756Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.243546Z","title":null,"venue":null,"work_id":"2a7dfd82-ae75-41da-8712-87d4edacb279","year":null},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.846726Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:abf5037723d6ecd1bacaeb37764c75d72e8ce010eb7ddc8d7cf12273a1db095c","observation_id":"4a37a47a-7458-4da5-87f0-7374f43f504d","resolution":{"observed_at":"2026-08-06T20:03:18.246286Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.236183Z","title":null,"venue":null,"work_id":"24158073-18bd-4b80-83dc-cb002330f8ed","year":null},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.852370Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:940d538544aeede7d554f4df4bcf23684d4bc5a0d627cab17b926217cc879a1c","observation_id":"ee997eda-f97a-411d-add7-3492446b1997","resolution":{"observed_at":"2026-08-06T20:03:18.238744Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.228735Z","title":null,"venue":null,"work_id":"cd902a65-a1ca-458e-be96-adbc473a9323","year":null},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:17.978850Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:77a309c3971adf0df25441ec2508bec3f478e205370a5aa4d780cdc28914c0fe","observation_id":"5e13959f-06c3-468c-90a2-24e9a228116f","resolution":{"observed_at":"2026-08-06T20:03:18.231361Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.221008Z","title":"Adam: A method for stochastic opti- mization,","venue":null,"work_id":"937cbf2b-e91d-448e-b96c-1e4809c3ebe9","year":2015},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:18.015268Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:2d77a58d827b92f9f511e41bfef859bed83d36c2138a70480bbc236d5e3953fb","observation_id":"1043fdb1-3323-4995-85ab-f70300f9fcd1","resolution":{"observed_at":"2026-08-06T20:03:18.223716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.212892Z","title":"Prosodic features con- trol by symbols as input of sequence-to-sequence acoustic mod- eling for neural TTS,","venue":null,"work_id":"dab5e020-ac0f-451d-9c5a-22b34a564e1f","year":2021},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:18.018239Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:3239cc6687ed8172a1850d4e75d0069f8c8f7d0e6c0e99a7935267975c580b6e","observation_id":"03c44fd4-d47f-4fac-88aa-b3e519a329b4","resolution":{"observed_at":"2026-08-06T20:03:18.216152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.196739Z","title":"ESPnet: End-to-end speech processing toolkit,","venue":null,"work_id":"23fe92a8-d5a3-4151-8290-c4d05e733af0","year":2018},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:18.023597Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:74d0f7f081ff2e1ee2a444c369efdccdac00884ca9afeb2efa6dea3072eac9e1","observation_id":"ed323aca-ccbe-48df-8dd5-3fdc755c5c3f","resolution":{"observed_at":"2026-08-06T20:03:18.199524Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.188598Z","title":null,"venue":null,"work_id":"48ebbbad-4189-4611-84e9-eda357d3af13","year":null},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:18.026215Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:6a31313a6e65305292181e45aad8db345a759d5a5fd3b81e97224ee629d708ef","observation_id":"7a77b45e-0bef-43fa-a63d-dde25d8ad60f","resolution":{"observed_at":"2026-08-06T20:03:18.191205Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.090218Z","title":null,"venue":null,"work_id":"86426d54-3b0c-4c69-8388-f482903544b9","year":null},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:18.028687Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:d72e654dc0b29c6edde87d432c1d9912a532d5fc93f1c6cb5da6abb97353a7a0","observation_id":"4a17447c-9180-42b4-a351-1ad51b06dae9","resolution":{"observed_at":"2026-08-06T20:03:18.092629Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:03:18.081893Z","title":"WORLD: A vocoder- based high-quality speech synthesis system for real-time appli- cations,","venue":null,"work_id":"dd8f8892-583f-4fd8-9329-61f250b22d7c","year":2016},"citing_paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T20:03:18.031484Z"},"links":{"citing_paper":"/paper/2507.03912"},"observation_digest":"sha256:8f6f52219a7b755949121b690c3307d4c360e3538948118d6198b3a0ecc960ff","observation_id":"566d1044-1005-4142-a880-81b694d682dd","resolution":{"observed_at":"2026-08-06T20:03:18.084938Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.03912","last_updated":"2025-07-05T05:56:49Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-12T04:05:18.280302Z","submitted_at":"2025-07-05T05:56:49Z","title":"Prosody Labeling with Phoneme-BERT and Speech Foundation Models"},"reference_resolution":{"displayed":53,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":8,"verified_exact":1,"verified_fuzzy":43},"total_outbound_references":53},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 53 of 53 outbound references and 1 inbound Pith citation observation for arXiv:2507.03912."}