{"as_of":"2026-08-07T08:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b30d1c30b2d42a7b8fc46dea68d1a3b18f9904f510b30c8eed41c4ded3b58a40","coverage":[{"denominator":40,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:21:15.102583Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:21:14.926845Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T03:16:19.153331Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.08530","snapshot_observed_at":"2026-08-06T18:21:14.926845Z","title":"MIDI-V ALLE: Im- proving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.926845Z"},"links":{"cited_paper":"/paper/2507.08530","citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:b23a94c6cb007f6861d789ad7c62850631e343faa59698722fe50876780f7f72","observation_id":"4dc0ed6d-f6a3-4563-a897-3a0ddb9912d9","resolution":{"observed_at":"2026-08-06T18:21:14.926845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"cited_work":{"arxiv_id":"2507.08530","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.08530","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MIDI-V ALLE: Improving expressive piano performance synthesis through neural codec language modelling","venue":null,"work_id":"a660858a-9d4a-4748-bd7e-ce4e7dd95450","year":2025},"citing_paper":{"arxiv_id":"2605.10281","last_updated":"2026-05-11T09:40:14Z","snapshot_observed_at":"2026-07-06T23:22:18.704329Z","submitted_at":"2026-05-11T09:40:14Z","title":"Drum Synthesis from Expressive Drum Grids via Neural Audio Codecs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T03:13:47.971429Z"},"links":{"cited_paper":"/paper/2507.08530","citing_paper":"/paper/2605.10281"},"observation_digest":"sha256:b605f12e7b8db8f77918d7dfa87e465a5965a2fb576807a4fad32100f6966f31","observation_id":"3e9631ad-6b7c-4667-a14e-9888b00353a3","resolution":{"observed_at":"2026-05-12T03:16:19.155184Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.08530/citation-record","integrity":"/paper/2507.08530/integrity","json":"/paper/2507.08530/citation-record.json","paper":"/paper/2507.08530"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.08530","snapshot_observed_at":"2026-08-06T18:21:14.926845Z","title":"MIDI-V ALLE: Im- proving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.926845Z"},"links":{"cited_paper":"/paper/2507.08530","citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:b23a94c6cb007f6861d789ad7c62850631e343faa59698722fe50876780f7f72","observation_id":"4dc0ed6d-f6a3-4563-a897-3a0ddb9912d9","resolution":{"observed_at":"2026-08-06T18:21:14.926845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.833587Z","title":"These TTS-inspired models typically process piano performance MIDIs as pi- ano rolls for audio synthesis","venue":null,"work_id":"66f69933-e342-46f1-aa41-639750ecd101","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.932050Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:0d378b86e6efd7ab94724d4898a86a712ff25265fbe7c62de714ca6d56da97cf","observation_id":"adc50779-db5a-4d13-b79b-f53a95f4dfe7","resolution":{"observed_at":"2026-08-06T18:21:15.839116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.818200Z","title":null,"venue":null,"work_id":"c54a14d0-5de7-4ef8-b6db-c932864fc453","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.936140Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:c08897b39fa741f70fb7328bb7ba4de793067db0f9cf5d9eed8780a4ef034af6","observation_id":"cc66a967-88e0-41f2-af75-4355aec2f94d","resolution":{"observed_at":"2026-08-06T18:21:15.822756Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.802206Z","title":"A total of 8,825 perfor- mance recordings were selected and split into training, val- idation, and test sets in an 8:1:1 ratio","venue":null,"work_id":"d4b531df-b505-4e09-9df2-c6e9283ef432","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.940657Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:ee3635291f3516978d8dacbf6fe004e03e8628ca40a61a3aaac74dbd8fcdabcf","observation_id":"eb9e0395-746b-45ca-afa9-b6fd87f4a6a7","resolution":{"observed_at":"2026-08-06T18:21:15.808049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.787455Z","title":"FAD measures the percep- tual quality and realism of generated audio by comparing it to reference performances using embeddings extracted from Piano-Encodec","venue":null,"work_id":"107be66e-7486-4188-84c0-a2f1e241497c","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.945077Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:bd7b24e93777e54c16deeab40fd032331e62c72e4ca04520b0edb0ac75cfb6a0","observation_id":"27534c76-e234-4e62-84f5-2a234bdac3b1","resolution":{"observed_at":"2026-08-06T18:21:15.792345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.772275Z","title":"In addition, Piano-Encodec achieves high-fidelity reconstruction of human performances, with much lower FAD, spectrogram, and chroma distortions than generative models","venue":null,"work_id":"8f3c8bb8-3232-4a0f-846d-45d7e1614a3b","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.949035Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:8ce99f944df469d3036713929f47a51a84c2013ea12607de11598f6eeb2ded0b","observation_id":"c2412f64-3be6-4e97-a5c9-07d5832c08d8","resolution":{"observed_at":"2026-08-06T18:21:15.777406Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.758370Z","title":null,"venue":null,"work_id":"81a261ac-7987-422c-a753-446150639f03","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.953434Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:aec83dd6fe5c1cb7f72e436c6b949e7e4b1ca2f66d6bf997ae803d7c001f23dc","observation_id":"2d1f7a27-5078-427c-8ab1-ea8b2983983e","resolution":{"observed_at":"2026-08-06T18:21:15.762884Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.742665Z","title":"Onderzoeksprogramma Artificiële Intelli- gentie (AI) Vlaanderen","venue":null,"work_id":"20783172-6c0a-4794-ac98-67bb238f3288","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.958039Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:8b0c6c48ad698ba758daa6f0c1eed9d8181dc22e889980621b3286edb0c1f06d","observation_id":"fda4af9b-ca63-4fbf-a324-b04038c80ea8","resolution":{"observed_at":"2026-08-06T18:21:15.747605Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.726873Z","title":"The datasets used in this study — ATEPP [8], Mae- stro [6], and Pijama [31] — contain audio recordings and corresponding MIDI annotations of piano performances","venue":null,"work_id":"b1b09581-7a31-4916-a989-4bdeec209a02","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.961825Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:ab5e2fc205d2d7bf5ff63f554c289184ae2ef59e8632fc744841ef04d2cfaa7a","observation_id":"d2c57a8b-2e56-4540-9ab4-adeafd1b3669","resolution":{"observed_at":"2026-08-06T18:21:15.732182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.712114Z","title":"MIDI-DDSP: Detailed control of musical per- formance via hierarchical modeling,","venue":null,"work_id":"a39f2cb7-3ae2-4aea-8598-c58334a7abf3","year":2022},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.965754Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:65ed564fc6dd590378eb1b06f81ded415f3d7d8a57a0a9bdd7142bd27197f749","observation_id":"4aa43205-f3d7-4862-b2a6-7e7580c5704a","resolution":{"observed_at":"2026-08-06T18:21:15.716742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.697714Z","title":"Deep performer: Score-to-audio music performance synthesis,","venue":null,"work_id":"51a90b0e-d747-479c-8997-5e9387ed550d","year":2022},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.969627Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:036d485434e577054e55dbd2f5dad541a6cadc3a85986d508162df090eadde82","observation_id":"f4f5dddb-1725-4c0c-bf4d-77f39716a44f","resolution":{"observed_at":"2026-08-06T18:21:15.702118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.681811Z","title":"Towards an integrated approach for expressive piano performance synthesis from music scores,","venue":null,"work_id":"b0ab1b8c-1f0a-4713-bab7-b4f6128c3b13","year":2025},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.973662Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:47905b5e3694a367026cf016ee1686acbf82391a75969196fed82cd58230c4e8","observation_id":"27764fd2-095c-4163-bfad-95a642a9755e","resolution":{"observed_at":"2026-08-06T18:21:15.687524Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.666084Z","title":"Text-to- speech synthesis techniques for midi-to-audio synthe- sis,","venue":null,"work_id":"ff8a0058-6400-48e1-afb3-71c3019382e7","year":2021},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.978451Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:35b8f5f9508ca71b4a18687304f84e8e06c5c93b0fac241cdc52627317c88ede","observation_id":"62644193-c50b-4dc6-9a8c-5342284a8c09","resolution":{"observed_at":"2026-08-06T18:21:15.671093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.652151Z","title":"Can knowledge of end-to-end text-to- speech models improve neural midi-to-audio synthesis systems?","venue":null,"work_id":"b57268e0-4eb0-415f-947b-8098e59505d1","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.982858Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:f691913728f8c3fdc1384e4664b8dcc6860f3d38de2d47c16987e516e20ffbf5","observation_id":"4dd5afd7-4896-4cf9-b881-257f344aa412","resolution":{"observed_at":"2026-08-06T18:21:15.656402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.637663Z","title":"Enabling factorized piano music modeling and generation with the MAE- STRO dataset,","venue":null,"work_id":"a0325215-1b61-4cd3-9ee7-469b1e94d7ff","year":2019},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.987761Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:dfd562a2a8b302611d9294e8333eed9fb663606cb6dccf37b65fbd70785be710","observation_id":"7572fad2-59eb-4d17-88fb-5a09824af55f","resolution":{"observed_at":"2026-08-06T18:21:15.642388Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.622363Z","title":"Neural codec language models are zero-shot text to speech synthesizers,","venue":null,"work_id":"433ee0b6-3bba-47ff-8977-83426028407a","year":2025},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.991502Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:bda574bfaa5cca6a07021fdfe29bc07aa5595c642cdfb6f9b2f949be3fd67eac","observation_id":"2cd6a077-228d-4003-bff8-099abee8182e","resolution":{"observed_at":"2026-08-06T18:21:15.627133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:14.995391Z","title":"ATEPP: A Dataset of Auto- matically Transcribed Expressive Piano Performance,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.995391Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:8edac17ae81ac8196f84c36779d4449d2b8930b8d4678cce529cb92f49ceabc4","observation_id":"5b9a38f2-2168-48cc-91bb-043c0c868a2b","resolution":{"observed_at":"2026-08-06T18:21:14.995391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.608074Z","title":"MusicBERT: Symbolic music understanding with large-scale pre-training,","venue":null,"work_id":"0b30d374-e937-4e2f-8554-f1b4288d44e5","year":2021},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.999201Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:ea5dc75b1ae4b941778318c89fc653ecf9d7fad7aff0edacc0b59d471d392522","observation_id":"ecfb31d1-da3c-4ae6-afca-123faa57f43d","resolution":{"observed_at":"2026-08-06T18:21:15.612528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.005210Z","title":"High fidelity neural audio compression,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.005210Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:dea7dc61f6c7beb4d8040e85cedfa7e442dbe3facdf3e000b0027cdea2bab7e0","observation_id":"50f4c473-84bf-4fe9-8a1c-2c05c206074e","resolution":{"observed_at":"2026-08-06T18:21:15.005210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.583072Z","title":"DDSP-Piano: a Neural Sound Synthesizer Informed by Instrument Knowledge,","venue":null,"work_id":"4fe56f85-6933-4266-8974-31620ec7373b","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.010760Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:e913c82875016ab4306427b154afe36a692b31cf396b08904ea49482b22d7aeb","observation_id":"8ba431e2-887b-4c82-8bad-af5912692511","resolution":{"observed_at":"2026-08-06T18:21:15.587666Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.569302Z","title":"Neural speech synthesis with transformer network,","venue":null,"work_id":"012efcf0-5c2a-4061-be8c-d075621f66a0","year":2019},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.016055Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:984837c85eb2bd638721ba13786d0365599c5580e2f6e656ef752553249f5e16","observation_id":"c66a0b4f-0232-427d-ab43-b2d351839119","resolution":{"observed_at":"2026-08-06T18:21:15.573259Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.555610Z","title":"Fastspeech: fast, robust and control- lable text to speech,","venue":null,"work_id":"f6973b36-a19b-4b2b-9c35-5104526fd578","year":2019},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.021278Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:1b1800b7514e96107c79060e583f4d00cc8848b9f1e1a55efda21177ee4bedcd","observation_id":"c57e03a9-6f1b-475d-a133-1114f1828a73","resolution":{"observed_at":"2026-08-06T18:21:15.559542Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.540798Z","title":"Hifi-gan: Generative ad- versarial networks for efficient and high fidelity speech synthesis,","venue":null,"work_id":"45be15dc-c8f4-41f5-a063-01c6625c9a06","year":2020},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.025672Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:2bc2543506401c3b256ceffa112f93e88b4df506041a778bf0f70b96ded91557","observation_id":"a50a33c3-6277-42bf-b1dc-9523d6db5521","resolution":{"observed_at":"2026-08-06T18:21:15.545553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.526360Z","title":"Reconstructing human expressiveness in piano performances with a transformer network,","venue":null,"work_id":"f2c8fffa-c0b4-46bd-896e-652ea37942bd","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.030143Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:426a9741a189dbbd2650cb406fd57b98a963178a1d0a3bf040b266ef56df1abb","observation_id":"6697cc0b-7c68-44f0-b26c-20a6682421d6","resolution":{"observed_at":"2026-08-06T18:21:15.530918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.512546Z","title":"Scoreperformer: Expressive piano performance rendering with fine-grained con- trol","venue":null,"work_id":"c3a81402-7f75-4bb5-88fa-c9d357824e7b","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.034566Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:f1ec71f22dac726bf556dc61025741e45804a281f140d7843fb7d7929fec47b2","observation_id":"67f07ad6-6521-427f-9cdf-d152cf24c606","resolution":{"observed_at":"2026-08-06T18:21:15.516875Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.498891Z","title":"Expressive Piano Performance Rendering from Unpaired Data,","venue":null,"work_id":"a21d4d6e-e01e-43c2-a74c-9133d88213ef","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.039254Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:dba839578a11ff130c1ddbd939ed183f12c4c9b30435bc94f195d14299e90638","observation_id":"76354753-295a-401e-9f3e-82ceace3621d","resolution":{"observed_at":"2026-08-06T18:21:15.503199Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.484161Z","title":"Vir- tuosonet: A hierarchical rnn-based system for model- ing expressive piano performance,","venue":null,"work_id":"ef128911-e295-480a-8b91-865ed5b8e069","year":2019},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.044403Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:b0a442d9be771175bcefda77de975edb81989d204501451530ef4b1ff33b15d7","observation_id":"7271a73b-0ec9-4782-9e04-ae7946876da8","resolution":{"observed_at":"2026-08-06T18:21:15.488767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.469398Z","title":"Dexter: Learning and controlling performance expression with diffusion models,","venue":null,"work_id":"070684b7-59ae-4c65-a183-8670d1f0cb66","year":2024},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.049249Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:930d41a0ef5a0749c336065c2e21993161a8ce5bc2e2f789cc0b907d200caa5c","observation_id":"0618346e-3744-4a64-ad42-61e1367d43eb","resolution":{"observed_at":"2026-08-06T18:21:15.473673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.455197Z","title":"Audiolm: a language modeling approach to audio generation,","venue":null,"work_id":"8f245ea2-018f-4a6d-a248-8e7c491cb678","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.053958Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:e938f703ec59f1879b1f7cf05c40f9984ce241d7e24ac9bb15a937c9aad2a7ad","observation_id":"8165f67f-55ba-4f53-b50c-38912f552028","resolution":{"observed_at":"2026-08-06T18:21:15.459335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.441244Z","title":"Simple and controllable music generation,","venue":null,"work_id":"3e47ac67-04d8-45a0-a444-57bf58c9000f","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.058705Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:c28b9a817d15be1e7ad217adf27ece72a4df231ecc05d1428746de07b29f2554","observation_id":"e65589ce-34d2-4df4-bb2d-70ad276aa15d","resolution":{"observed_at":"2026-08-06T18:21:15.445673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.11325","last_updated":"2023-01-26T18:58:53Z","snapshot_observed_at":"2026-07-06T14:45:00.730733Z","submitted_at":"2023-01-26T18:58:53Z","title":"MusicLM: Generating Music From Text","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.11325","snapshot_observed_at":"2026-08-06T18:21:15.063353Z","title":"Musiclm: Generating music from text,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.063353Z"},"links":{"cited_paper":"/paper/2301.11325","citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:6f2fd10257fea89797ea61a91389b93db687ea5de23bc1f72b0a8c1564a58b24","observation_id":"74fb4272-7b1a-4ead-ae2b-a82c507d54d6","resolution":{"observed_at":"2026-08-06T18:21:15.063353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.067823Z","title":"Soundstream: An end-to-end neural audio codec,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.067823Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:32e12287e8bd48667db25c44559062e28bdd8b3369081d9eadb4721cd9a58ea8","observation_id":"9366f862-42ee-4345-91de-59f34ec595fd","resolution":{"observed_at":"2026-08-06T18:21:15.067823Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.426878Z","title":"Vector quantization,","venue":null,"work_id":"22c03af5-6ab6-4aff-ac94-cc9a56c590e6","year":1984},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.072429Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:7c5fbdd0db2ff9fcb96ee178760593b1b0daa6a80c5ada8552a716bdad4a9089","observation_id":"17d6f87b-6fab-462b-b978-5440b6a6b8f0","resolution":{"observed_at":"2026-08-06T18:21:15.430826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.412515Z","title":"Vall-e: A neural codec language model,","venue":null,"work_id":"651e44d4-8e58-4efc-abc1-571a16c97eec","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.076464Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:26009e6ce1ce789ee049af72268930d4ff4fa2ea6a1171d36cdce0707c326126","observation_id":"9cd49da2-a945-49a8-a346-574c74059506","resolution":{"observed_at":"2026-08-06T18:21:15.417057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.080822Z","title":"Compound word transformer: Learning to compose full-song music over dynamic directed hypergraphs,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.080822Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:a9853a61e5e16383e70f693edbe53fe60cb34ba8506c97623ef467738e8a0e6c","observation_id":"eb238fd6-9adb-4e49-abec-7ae058f79ec2","resolution":{"observed_at":"2026-08-06T18:21:15.080822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.085266Z","title":"Pop music transformer: Beat-based modeling and generation of expressive pop piano compositions,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.085266Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:b9978e62881da5b93e1af059e04aeeeb28d09c7ea20744d224fbc89de9ea1984","observation_id":"831b8157-ed7a-4379-aff7-c48e5905d790","resolution":{"observed_at":"2026-08-06T18:21:15.085266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.388623Z","title":"Zipformer: A faster and better encoder for automatic speech recognition,","venue":null,"work_id":"2b35154d-2784-47f0-98a1-f82e4f3be7e7","year":2024},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.090118Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:e3b2e72e29f0c6dcf01630dfa5c3a2e111898d2c6554b5b1b672d6c65f572006","observation_id":"4b62de7a-2490-43d3-b983-0e8cd0858188","resolution":{"observed_at":"2026-08-06T18:21:15.392896Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.373672Z","title":"Fréchet audio distance: A reference-free metric for evaluating music enhancement algorithms,","venue":null,"work_id":"d80b7440-a33d-43d7-adfe-fc3b684d6d69","year":2019},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.094347Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:fcb349cf5a48fd50f2e7507621d4c5f8acbf477db707af3e80ba32b067e3b8e8","observation_id":"de143774-171d-48c6-b390-dc9b22105721","resolution":{"observed_at":"2026-08-06T18:21:15.378393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.01616","last_updated":"2024-03-05T22:14:14Z","snapshot_observed_at":"2026-08-06T04:47:12.367496Z","submitted_at":"2023-11-02T21:58:55Z","title":"Adapting Frechet Audio Distance for Generative Music Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.01616","snapshot_observed_at":"2026-08-06T18:21:15.098354Z","title":"Adapting frechet audio distance for generative music evaluation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.098354Z"},"links":{"cited_paper":"/paper/2311.01616","citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:168fe524e84ed12e6734df2cd69c08d26a70c5d021e93242fd50ca1a8ca41e0c","observation_id":"53affdc1-ad3a-4d62-94e5-28260e76e17f","resolution":{"observed_at":"2026-08-06T18:21:15.098354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.356537Z","title":"Pijama: Pi- ano jazz with automatic midi annotations,","venue":null,"work_id":"1fb198cb-1bbf-47ed-886b-3b78ccfafcb1","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.102583Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:36730eaddf7e4e2a9153b836d4aafbac44aa75e1262de1aee62b086d3d96aa48","observation_id":"13f5afd0-517e-48dc-a994-faca780ef80e","resolution":{"observed_at":"2026-08-06T18:21:15.362987Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling"},"reference_resolution":{"displayed":40,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":9,"verified_exact":0,"verified_fuzzy":30},"total_outbound_references":40},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 40 of 40 outbound references and 2 inbound Pith citation observations for arXiv:2507.08530."}