{"as_of":"2026-08-17T09:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d41ed9746b331ca13a9822d0e28fdefe0008c03cf6bd67f739fce72d5ecd2976","coverage":[{"denominator":40,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:21:15.102583Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:21:14.926845Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T03:16:19.153331Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.08530","snapshot_observed_at":"2026-08-06T18:21:14.926845Z","title":"MIDI-V ALLE: Im- proving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.926845Z"},"links":{"cited_paper":"/paper/2507.08530","citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:b23a94c6cb007f6861d789ad7c62850631e343faa59698722fe50876780f7f72","observation_id":"4dc0ed6d-f6a3-4563-a897-3a0ddb9912d9","resolution":{"observed_at":"2026-08-06T18:21:14.926845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"cited_work":{"arxiv_id":"2507.08530","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.08530","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MIDI-V ALLE: Improving expressive piano performance synthesis through neural codec language modelling","venue":null,"work_id":"a660858a-9d4a-4748-bd7e-ce4e7dd95450","year":2025},"citing_paper":{"arxiv_id":"2605.10281","last_updated":"2026-05-11T09:40:14Z","snapshot_observed_at":"2026-08-13T17:48:58.631716Z","submitted_at":"2026-05-11T09:40:14Z","title":"Drum Synthesis from Expressive Drum Grids via Neural Audio Codecs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T03:13:47.971429Z"},"links":{"cited_paper":"/paper/2507.08530","citing_paper":"/paper/2605.10281"},"observation_digest":"sha256:ade6bd769d534696917350f29f7f54bbe870251191c933670ca0d4ed113c6377","observation_id":"3e9631ad-6b7c-4667-a14e-9888b00353a3","resolution":{"observed_at":"2026-05-12T03:16:19.155184Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.08530/citation-record","integrity":"/paper/2507.08530/integrity","json":"/paper/2507.08530/citation-record.json","paper":"/paper/2507.08530"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.08530","snapshot_observed_at":"2026-08-06T18:21:14.926845Z","title":"MIDI-V ALLE: Im- proving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.926845Z"},"links":{"cited_paper":"/paper/2507.08530","citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:b23a94c6cb007f6861d789ad7c62850631e343faa59698722fe50876780f7f72","observation_id":"4dc0ed6d-f6a3-4563-a897-3a0ddb9912d9","resolution":{"observed_at":"2026-08-06T18:21:14.926845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.833587Z","title":"These TTS-inspired models typically process piano performance MIDIs as pi- ano rolls for audio synthesis","venue":null,"work_id":"66f69933-e342-46f1-aa41-639750ecd101","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.932050Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:4dc0cfe38ff5f77faee9f214d82cddd410492cd3b5f5875b302302ceda269d27","observation_id":"adc50779-db5a-4d13-b79b-f53a95f4dfe7","resolution":{"observed_at":"2026-08-06T18:21:15.839116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.818200Z","title":null,"venue":null,"work_id":"c54a14d0-5de7-4ef8-b6db-c932864fc453","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.936140Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:44c4418155006ec753bb442d3e19f99fc29e5ec58640cd733517bb0df316ed44","observation_id":"cc66a967-88e0-41f2-af75-4355aec2f94d","resolution":{"observed_at":"2026-08-06T18:21:15.822756Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.802206Z","title":"A total of 8,825 perfor- mance recordings were selected and split into training, val- idation, and test sets in an 8:1:1 ratio","venue":null,"work_id":"d4b531df-b505-4e09-9df2-c6e9283ef432","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.940657Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:c1c9e8475462f39718382d3d40edabd3597d01c91bedf1b0fb4e124a77184735","observation_id":"eb9e0395-746b-45ca-afa9-b6fd87f4a6a7","resolution":{"observed_at":"2026-08-06T18:21:15.808049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.787455Z","title":"FAD measures the percep- tual quality and realism of generated audio by comparing it to reference performances using embeddings extracted from Piano-Encodec","venue":null,"work_id":"107be66e-7486-4188-84c0-a2f1e241497c","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.945077Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:c1276d1b7cabe9fbd532d0e5006a346b809a4b7fd3edf39f105c906648a9a1c4","observation_id":"27534c76-e234-4e62-84f5-2a234bdac3b1","resolution":{"observed_at":"2026-08-06T18:21:15.792345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.772275Z","title":"In addition, Piano-Encodec achieves high-fidelity reconstruction of human performances, with much lower FAD, spectrogram, and chroma distortions than generative models","venue":null,"work_id":"8f3c8bb8-3232-4a0f-846d-45d7e1614a3b","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.949035Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:68c816f6cd93e77245a653b6483a71249292c073fd9ff9ac76f4de38feadddb0","observation_id":"c2412f64-3be6-4e97-a5c9-07d5832c08d8","resolution":{"observed_at":"2026-08-06T18:21:15.777406Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.758370Z","title":null,"venue":null,"work_id":"81a261ac-7987-422c-a753-446150639f03","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.953434Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:ae7b6e0fbe56af9e30618df6997b5737599288c396410a1e7b5015c8e19e8a07","observation_id":"2d1f7a27-5078-427c-8ab1-ea8b2983983e","resolution":{"observed_at":"2026-08-06T18:21:15.762884Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.742665Z","title":"Onderzoeksprogramma Artificiële Intelli- gentie (AI) Vlaanderen","venue":null,"work_id":"20783172-6c0a-4794-ac98-67bb238f3288","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.958039Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:a6c5cc321e3fc7ed1d5007dd6298d8b76d88f917e111d00552b2ccb0eaf6db5b","observation_id":"fda4af9b-ca63-4fbf-a324-b04038c80ea8","resolution":{"observed_at":"2026-08-06T18:21:15.747605Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.726873Z","title":"The datasets used in this study — ATEPP [8], Mae- stro [6], and Pijama [31] — contain audio recordings and corresponding MIDI annotations of piano performances","venue":null,"work_id":"b1b09581-7a31-4916-a989-4bdeec209a02","year":null},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.961825Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:1e92dac393d20a9b4b1b81d36efebc8e878f2d530e6f4e37b56e08fd35c15d58","observation_id":"d2c57a8b-2e56-4540-9ab4-adeafd1b3669","resolution":{"observed_at":"2026-08-06T18:21:15.732182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.712114Z","title":"MIDI-DDSP: Detailed control of musical per- formance via hierarchical modeling,","venue":null,"work_id":"a39f2cb7-3ae2-4aea-8598-c58334a7abf3","year":2022},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.965754Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:5eeb1621095ed7dc7a888a6c85efebc72b037a50bd1f11e488a8f3b2a099b26f","observation_id":"4aa43205-f3d7-4862-b2a6-7e7580c5704a","resolution":{"observed_at":"2026-08-06T18:21:15.716742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.697714Z","title":"Deep performer: Score-to-audio music performance synthesis,","venue":null,"work_id":"51a90b0e-d747-479c-8997-5e9387ed550d","year":2022},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.969627Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:03debdef04083bc6aca58f2417bddebd51c790acbfc2e312e59bec4d72a4a54b","observation_id":"f4f5dddb-1725-4c0c-bf4d-77f39716a44f","resolution":{"observed_at":"2026-08-06T18:21:15.702118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.681811Z","title":"Towards an integrated approach for expressive piano performance synthesis from music scores,","venue":null,"work_id":"b0ab1b8c-1f0a-4713-bab7-b4f6128c3b13","year":2025},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.973662Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:8cc9a22beb3f21c2ab1d084910532a6f3384c850ec6eb795529cc2e3211e1124","observation_id":"27764fd2-095c-4163-bfad-95a642a9755e","resolution":{"observed_at":"2026-08-06T18:21:15.687524Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.666084Z","title":"Text-to- speech synthesis techniques for midi-to-audio synthe- sis,","venue":null,"work_id":"ff8a0058-6400-48e1-afb3-71c3019382e7","year":2021},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.978451Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:82601591bd2dcecb136223e15940b78688ef93dc5ea9ea9d287e9b4e523468f9","observation_id":"62644193-c50b-4dc6-9a8c-5342284a8c09","resolution":{"observed_at":"2026-08-06T18:21:15.671093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.652151Z","title":"Can knowledge of end-to-end text-to- speech models improve neural midi-to-audio synthesis systems?","venue":null,"work_id":"b57268e0-4eb0-415f-947b-8098e59505d1","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.982858Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:076fc09067a1eca59eedaab7a106fc1f3264ec3d3b7112c8c596a53ed9a21093","observation_id":"4dd5afd7-4896-4cf9-b881-257f344aa412","resolution":{"observed_at":"2026-08-06T18:21:15.656402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.637663Z","title":"Enabling factorized piano music modeling and generation with the MAE- STRO dataset,","venue":null,"work_id":"a0325215-1b61-4cd3-9ee7-469b1e94d7ff","year":2019},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.987761Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:46e5ec988768f52ffbd9be4a0a968947e8ce82bc8c1078e1322d867eb1b7da22","observation_id":"7572fad2-59eb-4d17-88fb-5a09824af55f","resolution":{"observed_at":"2026-08-06T18:21:15.642388Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.622363Z","title":"Neural codec language models are zero-shot text to speech synthesizers,","venue":null,"work_id":"433ee0b6-3bba-47ff-8977-83426028407a","year":2025},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.991502Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:443e886ca09109c6ba8d6f10087d1cd7b7dcf9f51a42a585d6e795be6e9ca5bb","observation_id":"2cd6a077-228d-4003-bff8-099abee8182e","resolution":{"observed_at":"2026-08-06T18:21:15.627133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:14.995391Z","title":"ATEPP: A Dataset of Auto- matically Transcribed Expressive Piano Performance,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.995391Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:8edac17ae81ac8196f84c36779d4449d2b8930b8d4678cce529cb92f49ceabc4","observation_id":"5b9a38f2-2168-48cc-91bb-043c0c868a2b","resolution":{"observed_at":"2026-08-06T18:21:14.995391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.608074Z","title":"MusicBERT: Symbolic music understanding with large-scale pre-training,","venue":null,"work_id":"0b30d374-e937-4e2f-8554-f1b4288d44e5","year":2021},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:14.999201Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:ca15c51e1d3be8dab1fa692fb4af96c3938a145a7053e650d4a04a76092f76cd","observation_id":"ecfb31d1-da3c-4ae6-afca-123faa57f43d","resolution":{"observed_at":"2026-08-06T18:21:15.612528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.005210Z","title":"High fidelity neural audio compression,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.005210Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:dea7dc61f6c7beb4d8040e85cedfa7e442dbe3facdf3e000b0027cdea2bab7e0","observation_id":"50f4c473-84bf-4fe9-8a1c-2c05c206074e","resolution":{"observed_at":"2026-08-06T18:21:15.005210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.583072Z","title":"DDSP-Piano: a Neural Sound Synthesizer Informed by Instrument Knowledge,","venue":null,"work_id":"4fe56f85-6933-4266-8974-31620ec7373b","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.010760Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:9fab78868ed97ef15ef4a2b863d0bf191283144c9b6adf3662595702b8dbbb98","observation_id":"8ba431e2-887b-4c82-8bad-af5912692511","resolution":{"observed_at":"2026-08-06T18:21:15.587666Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.569302Z","title":"Neural speech synthesis with transformer network,","venue":null,"work_id":"012efcf0-5c2a-4061-be8c-d075621f66a0","year":2019},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.016055Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:1ce6f67745884f1640287b5f655aa8ba14a41e24603a0a398076e0c49bd91f71","observation_id":"c66a0b4f-0232-427d-ab43-b2d351839119","resolution":{"observed_at":"2026-08-06T18:21:15.573259Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.555610Z","title":"Fastspeech: fast, robust and control- lable text to speech,","venue":null,"work_id":"f6973b36-a19b-4b2b-9c35-5104526fd578","year":2019},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.021278Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:fd89e50ec2014423b0c50be73f18b72f449eef6b9b737efcf705c581735f65a9","observation_id":"c57e03a9-6f1b-475d-a133-1114f1828a73","resolution":{"observed_at":"2026-08-06T18:21:15.559542Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.540798Z","title":"Hifi-gan: Generative ad- versarial networks for efficient and high fidelity speech synthesis,","venue":null,"work_id":"45be15dc-c8f4-41f5-a063-01c6625c9a06","year":2020},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.025672Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:ed6a1f9ed8128e6cdb637611e6e92d2d092bc1731ab02a4f5330796a1ca47a12","observation_id":"a50a33c3-6277-42bf-b1dc-9523d6db5521","resolution":{"observed_at":"2026-08-06T18:21:15.545553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.526360Z","title":"Reconstructing human expressiveness in piano performances with a transformer network,","venue":null,"work_id":"f2c8fffa-c0b4-46bd-896e-652ea37942bd","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.030143Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:53c271de850954bcacb15fe4e85441113c17ff24ca3e790100db4e0c4752fd0a","observation_id":"6697cc0b-7c68-44f0-b26c-20a6682421d6","resolution":{"observed_at":"2026-08-06T18:21:15.530918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.512546Z","title":"Scoreperformer: Expressive piano performance rendering with fine-grained con- trol","venue":null,"work_id":"c3a81402-7f75-4bb5-88fa-c9d357824e7b","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.034566Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:ad689b9be0d1f2bd517c66c4ff877159bae671a001de78294ad9c026a0957c8d","observation_id":"67f07ad6-6521-427f-9cdf-d152cf24c606","resolution":{"observed_at":"2026-08-06T18:21:15.516875Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.498891Z","title":"Expressive Piano Performance Rendering from Unpaired Data,","venue":null,"work_id":"a21d4d6e-e01e-43c2-a74c-9133d88213ef","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.039254Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:ad7abd6f427dab02b4b7449de91271bf5ad478c9756cbfcd336f0e70e488a716","observation_id":"76354753-295a-401e-9f3e-82ceace3621d","resolution":{"observed_at":"2026-08-06T18:21:15.503199Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.484161Z","title":"Vir- tuosonet: A hierarchical rnn-based system for model- ing expressive piano performance,","venue":null,"work_id":"ef128911-e295-480a-8b91-865ed5b8e069","year":2019},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.044403Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:953c6b9b44e7d637783d21dc12769765cadc9c85aec72c32f07112d49b97828c","observation_id":"7271a73b-0ec9-4782-9e04-ae7946876da8","resolution":{"observed_at":"2026-08-06T18:21:15.488767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.469398Z","title":"Dexter: Learning and controlling performance expression with diffusion models,","venue":null,"work_id":"070684b7-59ae-4c65-a183-8670d1f0cb66","year":2024},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.049249Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:daf41e87216ce46381799395f4c92006c859614ee81adbd25bb1f9f0625e7553","observation_id":"0618346e-3744-4a64-ad42-61e1367d43eb","resolution":{"observed_at":"2026-08-06T18:21:15.473673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.455197Z","title":"Audiolm: a language modeling approach to audio generation,","venue":null,"work_id":"8f245ea2-018f-4a6d-a248-8e7c491cb678","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.053958Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:7118e1323fdb04d937ab29051bfeb003c69ffed4fc9ed057e4f0944f9038b654","observation_id":"8165f67f-55ba-4f53-b50c-38912f552028","resolution":{"observed_at":"2026-08-06T18:21:15.459335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.441244Z","title":"Simple and controllable music generation,","venue":null,"work_id":"3e47ac67-04d8-45a0-a444-57bf58c9000f","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.058705Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:a9cb47f4da42a26d98148e0247dbcd7cdd15e4d480e80799e27101c0a2fe4015","observation_id":"e65589ce-34d2-4df4-bb2d-70ad276aa15d","resolution":{"observed_at":"2026-08-06T18:21:15.445673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.11325","last_updated":"2023-01-26T18:58:53Z","snapshot_observed_at":"2026-08-17T00:33:41.369960Z","submitted_at":"2023-01-26T18:58:53Z","title":"MusicLM: Generating Music From Text","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.11325","snapshot_observed_at":"2026-08-06T18:21:15.063353Z","title":"Musiclm: Generating music from text,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.063353Z"},"links":{"cited_paper":"/paper/2301.11325","citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:560be8a92c2b8744e91f1bae911dd449633463c7360ff5272a663fb33448cef5","observation_id":"74fb4272-7b1a-4ead-ae2b-a82c507d54d6","resolution":{"observed_at":"2026-08-06T18:21:15.063353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.067823Z","title":"Soundstream: An end-to-end neural audio codec,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.067823Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:32e12287e8bd48667db25c44559062e28bdd8b3369081d9eadb4721cd9a58ea8","observation_id":"9366f862-42ee-4345-91de-59f34ec595fd","resolution":{"observed_at":"2026-08-06T18:21:15.067823Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.426878Z","title":"Vector quantization,","venue":null,"work_id":"22c03af5-6ab6-4aff-ac94-cc9a56c590e6","year":1984},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.072429Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:0059e16d2e859c50631648392710577da903a3605b36b0195a763c3cf42bdf37","observation_id":"17d6f87b-6fab-462b-b978-5440b6a6b8f0","resolution":{"observed_at":"2026-08-06T18:21:15.430826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.412515Z","title":"Vall-e: A neural codec language model,","venue":null,"work_id":"651e44d4-8e58-4efc-abc1-571a16c97eec","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.076464Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:52097f00410b81ba801d57bee4f81c61d56072b61ab00090b1f46f4cf8daa5b0","observation_id":"9cd49da2-a945-49a8-a346-574c74059506","resolution":{"observed_at":"2026-08-06T18:21:15.417057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.080822Z","title":"Compound word transformer: Learning to compose full-song music over dynamic directed hypergraphs,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.080822Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:a9853a61e5e16383e70f693edbe53fe60cb34ba8506c97623ef467738e8a0e6c","observation_id":"eb238fd6-9adb-4e49-abec-7ae058f79ec2","resolution":{"observed_at":"2026-08-06T18:21:15.080822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.085266Z","title":"Pop music transformer: Beat-based modeling and generation of expressive pop piano compositions,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.085266Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:b9978e62881da5b93e1af059e04aeeeb28d09c7ea20744d224fbc89de9ea1984","observation_id":"831b8157-ed7a-4379-aff7-c48e5905d790","resolution":{"observed_at":"2026-08-06T18:21:15.085266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.388623Z","title":"Zipformer: A faster and better encoder for automatic speech recognition,","venue":null,"work_id":"2b35154d-2784-47f0-98a1-f82e4f3be7e7","year":2024},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.090118Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:41a2051196a04479b27417ecfd5ede8307696310420a75e16d55d119c1a70196","observation_id":"4b62de7a-2490-43d3-b983-0e8cd0858188","resolution":{"observed_at":"2026-08-06T18:21:15.392896Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.373672Z","title":"Fréchet audio distance: A reference-free metric for evaluating music enhancement algorithms,","venue":null,"work_id":"d80b7440-a33d-43d7-adfe-fc3b684d6d69","year":2019},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.094347Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:b3d48692084799cc25bcb059e4feb69fdedf28c8f5df7ccf1dc843b3365998a9","observation_id":"de143774-171d-48c6-b390-dc9b22105721","resolution":{"observed_at":"2026-08-06T18:21:15.378393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.01616","last_updated":"2024-03-05T22:14:14Z","snapshot_observed_at":"2026-08-16T14:46:22.801988Z","submitted_at":"2023-11-02T21:58:55Z","title":"Adapting Frechet Audio Distance for Generative Music Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.01616","snapshot_observed_at":"2026-08-06T18:21:15.098354Z","title":"Adapting frechet audio distance for generative music evaluation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.098354Z"},"links":{"cited_paper":"/paper/2311.01616","citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:d4e655d62c9f03874b86fe137264054d22bd42d7ea5287d37e5c6b7acc7737f9","observation_id":"53affdc1-ad3a-4d62-94e5-28260e76e17f","resolution":{"observed_at":"2026-08-06T18:21:15.098354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:15.356537Z","title":"Pijama: Pi- ano jazz with automatic midi annotations,","venue":null,"work_id":"1fb198cb-1bbf-47ed-886b-3b78ccfafcb1","year":2023},"citing_paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T18:21:15.102583Z"},"links":{"citing_paper":"/paper/2507.08530"},"observation_digest":"sha256:8e5d80473b97c956885e9c8f9e77458ba69fe9134c9644d4c27212ef21086eda","observation_id":"13f5afd0-517e-48dc-a994-faca780ef80e","resolution":{"observed_at":"2026-08-06T18:21:15.362987Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.08530","last_updated":"2025-07-11T12:28:20Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-06T18:14:13.453174Z","submitted_at":"2025-07-11T12:28:20Z","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling"},"reference_resolution":{"displayed":40,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":9,"verified_exact":0,"verified_fuzzy":30},"total_outbound_references":40},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 40 of 40 outbound references and 2 inbound Pith citation observations for arXiv:2507.08530."}