{"as_of":"2026-08-20T15:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:55a353977ff99f6b22716157abc73d423ae35168446869e0b7ded0c4c447b3ed","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:57:08.122917Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T02:12:58.120295Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-11T02:17:46.305033Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"cited_work":{"arxiv_id":"2506.23009","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.23009","snapshot_observed_at":"2026-07-11T02:17:46.305033Z","title":"arXiv preprint arXiv:2506.23009 (2025)","venue":"cs.CV","work_id":"5de567d0-d759-4229-8585-c008c17da27a","year":2025},"citing_paper":{"arxiv_id":"2604.20719","last_updated":"2026-04-22T16:06:48Z","snapshot_observed_at":"2026-08-13T11:58:36.990456Z","submitted_at":"2026-04-22T16:06:48Z","title":"ONOTE: Benchmarking Omnimodal Notation Processing for Expert-level Music Intelligence","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-09T23:03:01.356594Z"},"links":{"cited_paper":"/paper/2506.23009","citing_paper":"/paper/2604.20719"},"observation_digest":"sha256:a012db9d57403ff3456dabe57e1dd8a94a2bc5aa2b9c9d3aa8e75d115e9eface","observation_id":"d96761d1-6795-4eab-9920-7f5e5c1453f1","resolution":{"observed_at":"2026-05-09T23:04:17.568533Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"cited_work":{"arxiv_id":"2506.23009","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.23009","snapshot_observed_at":"2026-07-11T02:17:46.305033Z","title":"arXiv preprint arXiv:2506.23009 (2025)","venue":"cs.CV","work_id":"5de567d0-d759-4229-8585-c008c17da27a","year":2025},"citing_paper":{"arxiv_id":"2605.22255","last_updated":"2026-05-28T16:18:50Z","snapshot_observed_at":"2026-08-18T12:36:29.685685Z","submitted_at":"2026-05-21T09:59:59Z","title":"Direct content-based retrieval from music scores images","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-22T07:21:33.390581Z"},"links":{"cited_paper":"/paper/2506.23009","citing_paper":"/paper/2605.22255"},"observation_digest":"sha256:efc3f33dcc4b068304368ab6dc2dcb7174eb89bec6cf305826bd27e12aeee431","observation_id":"d3580fe4-5cda-4565-a885-f9ca87d51454","resolution":{"observed_at":"2026-05-22T07:24:43.335938Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"cited_work":{"arxiv_id":"2506.23009","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.23009","snapshot_observed_at":"2026-07-11T02:17:46.305033Z","title":"arXiv preprint arXiv:2506.23009 (2025)","venue":"cs.CV","work_id":"5de567d0-d759-4229-8585-c008c17da27a","year":2025},"citing_paper":{"arxiv_id":"2605.22255","last_updated":"2026-05-28T16:18:50Z","snapshot_observed_at":"2026-08-18T12:36:29.685685Z","submitted_at":"2026-05-21T09:59:59Z","title":"Direct content-based retrieval from music scores images","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-30T17:23:39.285332Z"},"links":{"cited_paper":"/paper/2506.23009","citing_paper":"/paper/2605.22255"},"observation_digest":"sha256:7dc496516301dc37fe287422dffc494d04c6c1ab959718ac089c21ce0348aeb8","observation_id":"eb9799a3-f455-45e9-8103-3613ff422f8a","resolution":{"observed_at":"2026-06-30T17:24:56.953564Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"cited_work":{"arxiv_id":"2506.23009","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.23009","snapshot_observed_at":"2026-07-11T02:17:46.305033Z","title":"arXiv preprint arXiv:2506.23009 (2025)","venue":"cs.CV","work_id":"5de567d0-d759-4229-8585-c008c17da27a","year":2025},"citing_paper":{"arxiv_id":"2607.05769","last_updated":"2026-07-07T02:47:13Z","snapshot_observed_at":"2026-08-19T11:44:13.303428Z","submitted_at":"2026-07-07T02:47:13Z","title":"LEGATO 2: Toward Multimodal Sheet Music Recognition and Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-11T02:12:58.120295Z"},"links":{"cited_paper":"/paper/2506.23009","citing_paper":"/paper/2607.05769"},"observation_digest":"sha256:2ee24e13dec1b0d3dbd2f1c0f6479b386d5aa0d38fd7e204648d9bad650e30dc","observation_id":"9c192de3-7c3e-4328-ba8d-4758b25ca7c8","resolution":{"observed_at":"2026-07-11T02:17:46.326748Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"cited_work":{"arxiv_id":"2506.23009","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.23009","snapshot_observed_at":"2026-07-11T02:17:46.305033Z","title":"arXiv preprint arXiv:2506.23009 (2025)","venue":"cs.CV","work_id":"5de567d0-d759-4229-8585-c008c17da27a","year":2025},"citing_paper":{"arxiv_id":"2607.06015","last_updated":"2026-07-07T08:57:34Z","snapshot_observed_at":"2026-08-18T01:53:38.537475Z","submitted_at":"2026-07-07T08:57:34Z","title":"Music I Care About: Automated Multimodal Benchmarking of LLM Music Perception Skills on (Almost) Any Music","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-08T19:28:58.523932Z"},"links":{"cited_paper":"/paper/2506.23009","citing_paper":"/paper/2607.06015"},"observation_digest":"sha256:65b2cb741fe890f8ba259f05510ecc8dde3e8be673e4e413f231d32db5feb058","observation_id":"297aee09-8d2f-435a-a6c9-577e60120dae","resolution":{"observed_at":"2026-07-08T19:35:32.938812Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.23009/citation-record","integrity":"/paper/2506.23009/integrity","json":"/paper/2506.23009/citation-record.json","paper":"/paper/2506.23009"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:14.194805Z","title":"Acrobat AI Assistant, 2024","venue":null,"work_id":"71667200-43a9-466a-a46a-d00a653b62a0","year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.029002Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:0c61cdc3b3d1670cecfb164cbe85986f63cd491d1341ebfded036ca94aeeb470","observation_id":"f5cc43c0-2f24-47c8-9aa3-c5a96aeb3049","resolution":{"observed_at":"2026-08-06T21:57:14.300132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:03.094544Z","title":"Mmmu: A mas- sive multi-discipline multimodal understanding and reasoning benchmark for expert agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.094544Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:8d6e4b9c27a06f39e6920f861afd9d41c01a71ccc382ea05d42e3effbcbca974","observation_id":"940c5739-7079-4665-a70c-b0ff2e4571b7","resolution":{"observed_at":"2026-08-06T21:57:03.094544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:13.992829Z","title":"The basics of reading music","venue":null,"work_id":"fe8e1757-4387-4cac-988e-8210e1b8f74d","year":2015},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.170985Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:490edf7434e2095f1eaef67cd34d44c4126f04b6d4ef7a70af45e4b78b7d89a7","observation_id":"d0e24319-069e-4c41-8abe-6d1fc94fce4e","resolution":{"observed_at":"2026-08-06T21:57:14.081606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:13.762498Z","title":"Reading sheet music facilitates sensorimotor mu- desynchronization in musicians","venue":null,"work_id":"cfec6b65-f75a-40d0-a9a0-ab371606cdf5","year":2011},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.243569Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:dd196a64b9b8a9cca0f663ee6039801f861fa896646d6a43f10119f9be21a5f2","observation_id":"ccb2d247-30ec-4ed1-99b3-f219450ede58","resolution":{"observed_at":"2026-08-06T21:57:13.876029Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.07885","last_updated":"2020-06-22T16:33:59Z","snapshot_observed_at":"2026-08-18T01:53:39.029713Z","submitted_at":"2020-06-14T12:40:17Z","title":"Optical Music Recognition: State of the Art and Major Challenges","version":2},"cited_work":{"arxiv_id":"2006.07885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.07885","snapshot_observed_at":"2026-08-06T21:57:08.867615Z","title":"Optical Music Recognition: State of the Art and Major Challenges","venue":"cs.CV","work_id":"39c53193-b76f-45fa-aaf0-512225641bc8","year":2020},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.307574Z"},"links":{"cited_paper":"/paper/2006.07885","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:78cbf3b273dc94d92ff43234a22ce31985eed1ef8edc83461213d59d070f384f","observation_id":"444834fa-8cde-4875-89f3-9e5004298c0b","resolution":{"observed_at":"2026-08-06T21:57:08.894312Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:13.505963Z","title":"Understanding optical music recognition.ACM Computing Surveys (CSUR), 53(4):1–35, 2020","venue":null,"work_id":"a8249f75-cae6-4f51-864d-1601367c1079","year":2020},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.381957Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:f2fbeb089b47251d94b74ee8a80569262d25dfe990cb7f744055c60db0c0fafd","observation_id":"b20ed1a5-0a3d-49c0-87c4-c2fcf0ba7def","resolution":{"observed_at":"2026-08-06T21:57:13.647645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:13.376174Z","title":"Optical music recognition: state-of-the-art and open issues","venue":null,"work_id":"90b80fde-d070-49cd-9d7a-6948baf413b4","year":2012},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.441555Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:a3b534012b9df8438345b63e4c8b8b5e0e358ed1a4c4c4daded0a3bcd96e781a","observation_id":"0ad181d0-58c5-4a4d-a5c6-6942d6910ebe","resolution":{"observed_at":"2026-08-06T21:57:13.437024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:13.186284Z","title":"The challenge of opti- cal music recognition","venue":null,"work_id":"2dd49373-66d6-470e-bb8c-b06e5d4ec3bd","year":2001},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.530806Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:43381cbde83357be888c09633bf044144afaf33ade323e994f59223d4352ad14","observation_id":"3cedc974-803e-4108-9133-15be13d1487a","resolution":{"observed_at":"2026-08-06T21:57:13.262422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:13.004193Z","title":"Optical music recognition using pro- jections","venue":null,"work_id":"3864ded2-0d96-4d01-8e88-14c292d7144a","year":1988},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.612404Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:23b649a84527df6ec1f3c67e650c5b662755c4e8663d7b87e246791aeeeea6a7","observation_id":"a453cfc9-af66-446f-9777-0952ee80cedc","resolution":{"observed_at":"2026-08-06T21:57:13.096993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:03.704090Z","title":"Gui agents: A survey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.704090Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:849f0a937cba9df6328157bbbdd697fde0ea1a1f4f233fe0189b5e39515fc3f2","observation_id":"60d3e49a-7c68-4e09-9e4d-418e773af6dc","resolution":{"observed_at":"2026-08-06T21:57:03.704090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:12.764356Z","title":"Natural language understand- ing and inference with mllm in visual question answer- ing: A survey","venue":null,"work_id":"16b0477b-4ac0-4a43-b8e1-5a59069b3099","year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.796792Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:0a9e20f1f297b246d9c4ec161606c7d765dffb599129d50834f040005361e381","observation_id":"ea595109-5701-468c-b1e6-3ea06fd2ff0f","resolution":{"observed_at":"2026-08-06T21:57:12.868582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:12.568695Z","title":"Internvl: Scal- ing up vision foundation models and aligning for generic visual-linguistic tasks","venue":null,"work_id":"d4c2d602-30fe-4e5f-ba74-094b35990184","year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.801270Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:a75f2acfd726d81a1079cd2e0bdc3c42e935186a323ab799255071670da53549","observation_id":"5a9369b0-1cf4-4a8c-be2d-cce557455e5b","resolution":{"observed_at":"2026-08-06T21:57:12.665173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10727","last_updated":"2025-04-11T10:01:13Z","snapshot_observed_at":"2026-08-19T01:21:22.593463Z","submitted_at":"2024-01-19T14:44:37Z","title":"MLLM-Tool: A Multimodal Large Language Model For Tool Agent Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.10727","snapshot_observed_at":"2026-08-06T21:57:03.931871Z","title":"Mllm-tool: A multimodal large language model for tool agent learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:03.931871Z"},"links":{"cited_paper":"/paper/2401.10727","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:d122a93a2377521b18934d815a271033943e9a87c6a06f6704183f2d98c70550","observation_id":"320ef79e-a083-4b28-85bf-72cc75c4383a","resolution":{"observed_at":"2026-08-06T21:57:03.931871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:12.381531Z","title":"Mllm-as-a-judge: Assessing multimodal llm-as-a-judge with vision- language benchmark","venue":null,"work_id":"4ef6bf84-c144-4dad-aed2-ae235d906c77","year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:04.036358Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:f491aa3508ce93300b7bbbb87501da2b999b6f0506987f5eed5a084b3a243b5f","observation_id":"aa2550e8-38f8-4dd0-8235-07e20b3ea904","resolution":{"observed_at":"2026-08-06T21:57:12.448383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.09941","last_updated":"2020-10-15T14:21:53Z","snapshot_observed_at":"2026-08-17T14:49:36.972548Z","submitted_at":"2020-09-21T14:57:18Z","title":"PP-OCR: A Practical Ultra Lightweight OCR System","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.09941","snapshot_observed_at":"2026-08-06T21:57:04.136009Z","title":"Pp-ocr: A practi- cal ultra lightweight ocr system","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:04.136009Z"},"links":{"cited_paper":"/paper/2009.09941","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:56fcaeb325b321d3f9c104fc9672fc9e785e8234bddc616782726be46e61f465","observation_id":"c8f6ce39-1320-4708-a18d-31ee29f69ce0","resolution":{"observed_at":"2026-08-06T21:57:04.136009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:12.211777Z","title":"Tex- tocr: Towards large-scale end-to-end reasoning for arbitrary-shaped scene text","venue":null,"work_id":"a1cfc72d-9388-4f79-bbbe-30790a354c66","year":2021},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:04.213329Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:44ec7e74afea25075985b826ecd1dbc9fe6ea3a1629bf640b7a9d257644f9b72","observation_id":"8a612c9c-cc42-4bb0-a4a4-3444743860b3","resolution":{"observed_at":"2026-08-06T21:57:12.295573Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:12.017264Z","title":"CVC-MUSCIMA: A ground-truth of hand- written music score images for writer identification and staff removal","venue":null,"work_id":"fdd30b3c-fa91-4831-ae09-e6b37ea2899b","year":2012},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:04.318571Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:bc44593d19e97ee4046561b8a7b696f7a8cd3f4af230081217df0f35a7baabf0","observation_id":"74b8b8f5-9138-4277-8ef6-e870f4d84a8b","resolution":{"observed_at":"2026-08-06T21:57:12.108776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15002","last_updated":"2024-09-16T11:38:10Z","snapshot_observed_at":"2026-08-16T13:23:31.353848Z","submitted_at":"2024-08-27T12:34:41Z","title":"Knowledge Discovery in Optical Music Recognition: Enhancing Information Retrieval with Instance Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15002","snapshot_observed_at":"2026-08-06T21:57:04.402673Z","title":"Knowledge dis- covery in optical music recognition: Enhancing infor- mation retrieval with instance segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:04.402673Z"},"links":{"cited_paper":"/paper/2408.15002","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:5b4612b8ea8644874713280d43604f753b479eb78b0868da2bf075814742e18f","observation_id":"bf1091cd-1469-437c-9c43-c6694acdb18c","resolution":{"observed_at":"2026-08-06T21:57:04.402673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:11.845387Z","title":"Deepscores-a dataset for segmentation, detection and classification of tiny objects","venue":null,"work_id":"adaf6223-9c20-4d5e-9420-5b0abc1f794d","year":2018},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:04.520127Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:e6f117845ef5f70acfd28b47dedd7f0bea9427448da4afc844f1264bc03bbe41","observation_id":"25e45c23-f6c6-4e85-8d72-e701d2888fbe","resolution":{"observed_at":"2026-08-06T21:57:11.927933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:11.628351Z","title":"End-to- end neural optical music recognition of monophonic scores","venue":null,"work_id":"4166ef8e-3df8-44b0-ad8b-165ef3fd7ca9","year":2018},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:04.652406Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:0d77240f89d21ee2782c19223c29df6ccc2988e62b501c11dd841431ab4732f6","observation_id":"c8cf186a-0bc7-471d-a910-eb80ebd255d3","resolution":{"observed_at":"2026-08-06T21:57:11.748658Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.07786","last_updated":"2021-07-16T09:24:58Z","snapshot_observed_at":"2026-08-18T02:56:20.382317Z","submitted_at":"2021-07-16T09:24:58Z","title":"DoReMi: First glance at a universal OMR dataset","version":1},"cited_work":{"arxiv_id":"2107.07786","doi":null,"metadata_source":"pith","pith_arxiv_id":"2107.07786","snapshot_observed_at":"2026-08-06T21:57:08.487997Z","title":"DoReMi: First glance at a universal OMR dataset","venue":"cs.IR","work_id":"087e0949-ff6a-4bea-9082-5bd45f70e61a","year":2021},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:04.841644Z"},"links":{"cited_paper":"/paper/2107.07786","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:efc0f74337e1ac8a126dd58d5b5399ea769535cd27a8215f1ff1352670ee2eb8","observation_id":"dde69ff4-d1c0-4e62-bb9f-2b8729f92b64","resolution":{"observed_at":"2026-08-06T21:57:08.584722Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:11.389920Z","title":"A uni- fied representation framework for the evaluation of optical music recognition systems","venue":null,"work_id":"e2521254-bdc5-4f8f-9c60-db5bf473d1a5","year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:04.953302Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:c164910a4caa189e5e511cd039dcb605a322ff38dfbaa855e577fee119eb2d77","observation_id":"58138aea-ac0a-4327-bed4-ac749362497e","resolution":{"observed_at":"2026-08-06T21:57:11.509970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:11.149644Z","title":"Practical end-to-end optical music recognition for pianoform music","venue":null,"work_id":"1db4d3ef-b418-458d-b5dd-f63b5c82c1e7","year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:05.038931Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:6a70f4c05a48f0ef15294bee5e1b35e43ca03420d5ab3edd9debd3881d0e5f82","observation_id":"b9ccfb9a-77e1-4966-b55f-2c25af092de3","resolution":{"observed_at":"2026-08-06T21:57:11.238511Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:10.929326Z","title":"Breezewhite/oemer: v0.1.7, October 2023","venue":null,"work_id":"750f5cf0-e6ad-4d7e-8f08-b42c9d646ad4","year":2023},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:05.155593Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:f96a5e88302028a010cb38a520b8b63292a8c07c2e06afa6ff75732e969f5f2c","observation_id":"0099d064-ce21-4194-ad6f-d8db0da76193","resolution":{"observed_at":"2026-08-06T21:57:11.038775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:10.742712Z","title":"Optical music recognition in manuscripts from the ricordi archive","venue":null,"work_id":"c331a6f7-22a8-452d-a3f7-6ca3612e9ad3","year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:05.249120Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:eb6589cba5456e57d4574b86ba6a739bd825be0675fcc518087f1c16fc7f38f3","observation_id":"acb50a2c-7375-4bd9-9de1-923c41045971","resolution":{"observed_at":"2026-08-06T21:57:10.853442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:10.508064Z","title":"Optical music recognition with convolutional sequence-to-sequence models","venue":null,"work_id":"b1985944-aba6-48f5-8ca3-2227ae523d40","year":2017},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:05.356627Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:141916f52599b4b8a57451c78d81d7901d2bc09dc00cd9997249ae1ea7a4e3e9","observation_id":"5d5ad310-4581-4137-910f-76be0515c404","resolution":{"observed_at":"2026-08-06T21:57:10.606216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:10.334126Z","title":"Tromr:transformer-based polyphonic optical music recognition","venue":null,"work_id":"af9c4d9a-c173-424e-a1fe-6bd1b53b5c58","year":2023},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:05.523218Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:a4c32ee7a64690ef33d9729c065d4d0bbbcb962b1442ae069c01e4517fd64159","observation_id":"3f4661b1-32d6-469b-a1fd-69aa8172e667","resolution":{"observed_at":"2026-08-06T21:57:10.417118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:10.170757Z","title":"Sheet music transformer: End-to-end optical music recognition beyond monophonic transcription, 2024","venue":null,"work_id":"78b15452-cd26-4cb3-bfe2-1466a4389b43","year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:05.682538Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:3914d2797b6d42aa8aa22976d090c9ef9e3a6540ae63f7215650e9b057e766f4","observation_id":"1f102df9-df64-4f5e-98b4-c40740b6ed78","resolution":{"observed_at":"2026-08-06T21:57:10.244977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12105","last_updated":"2025-06-27T08:39:52Z","snapshot_observed_at":"2026-08-17T03:05:28.054197Z","submitted_at":"2024-05-20T15:21:48Z","title":"End-to-End Full-Page Optical Music Recognition for Pianoform Sheet Music","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12105","snapshot_observed_at":"2026-08-06T21:57:05.787561Z","title":"Sheet music trans- former++: End-to-end full-page optical music recog- nition for pianoform sheet music","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:05.787561Z"},"links":{"cited_paper":"/paper/2405.12105","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:997006ce1b9ac9c3f6c31ba98cdeaa8e7c64245484f8af3ad8d1871f31ec11bb","observation_id":"e6bbf8a2-3f79-4115-9454-0853be89d164","resolution":{"observed_at":"2026-08-06T21:57:05.787561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.16153","last_updated":"2024-02-25T17:19:41Z","snapshot_observed_at":"2026-08-16T14:15:31.491085Z","submitted_at":"2024-02-25T17:19:41Z","title":"ChatMusician: Understanding and Generating Music Intrinsically with LLM","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.16153","snapshot_observed_at":"2026-08-06T21:57:05.952693Z","title":"Chatmusician: Under- standing and generating music intrinsically with llm","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:05.952693Z"},"links":{"cited_paper":"/paper/2402.16153","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:ae5f03285050b497aa1cb481b7e8d0b437121ecab4e1c924bff39a5b96d42ab2","observation_id":"b25e32c3-7c57-4510-a441-9fc3db13dafe","resolution":{"observed_at":"2026-08-06T21:57:05.952693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.11954","last_updated":"2023-10-25T13:34:13Z","snapshot_observed_at":"2026-08-18T01:37:43.115781Z","submitted_at":"2023-10-18T13:31:10Z","title":"MusicAgent: An AI Agent for Music Understanding and Generation with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.11954","snapshot_observed_at":"2026-08-06T21:57:06.012244Z","title":"Mu- sicagent: An ai agent for music understanding and generation with large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:06.012244Z"},"links":{"cited_paper":"/paper/2310.11954","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:a28640f053f9f1b0c68d87af7a2ebf7d514632c8848ff03165a71688a433ded5","observation_id":"14ff5879-dfb9-45d9-8db0-5ea7db605ac8","resolution":{"observed_at":"2026-08-06T21:57:06.012244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03555","last_updated":"2024-12-04T18:50:42Z","snapshot_observed_at":"2026-07-06T20:01:45.826971Z","submitted_at":"2024-12-04T18:50:42Z","title":"PaliGemma 2: A Family of Versatile VLMs for Transfer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.03555","snapshot_observed_at":"2026-08-06T21:57:06.098344Z","title":"Paligemma 2: A family of versatile vlms for transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:06.098344Z"},"links":{"cited_paper":"/paper/2412.03555","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:25d5c7ba2cec32a21a28302e8a22d1183dc1bdfe7cfbd04a43517a8ecab214f6","observation_id":"d66af441-d8e1-4675-89bf-06959580841b","resolution":{"observed_at":"2026-08-06T21:57:06.098344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-08-15T14:02:47.366139Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-06T21:57:06.179397Z","title":"Gpt-4o system card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:06.179397Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:7a3b60a47c3e39d9404c02b506edb194bd4346f11cec58603f68b9ed7f7981d0","observation_id":"df89a664-d190-445b-81c5-1049c1ca726c","resolution":{"observed_at":"2026-08-06T21:57:06.179397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-08-18T18:18:37.449517Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T21:57:06.325770Z","title":"Deepseek- v3 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:06.325770Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:30c9897dfa0786300f4419433667b4c84a1aafea4f8afd7156d5f8e2d60d8007","observation_id":"a7908338-7a3c-4c00-b9bc-dd65d225b05f","resolution":{"observed_at":"2026-08-06T21:57:06.325770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:10.024874Z","title":"Musical scales and the generalized circle of fifths","venue":null,"work_id":"2bf80f39-18bf-4451-9607-642f28435097","year":1986},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:06.425346Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:24d1f88856f7f810c9b9a42e951066828faddc373231f9d9bb97e59b4916f02e","observation_id":"369ae79e-0a88-43c2-9875-83f009103400","resolution":{"observed_at":"2026-08-06T21:57:10.090242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:09.848991Z","title":"MusiXTEX","venue":null,"work_id":"dcbf65c4-f752-4114-a504-93faa08f13db","year":1993},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:06.520548Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:c8a7788d1f44761b1ea43881f64b523cf1305938b97aa8cec27e27e44b24de18","observation_id":"0c53bfd0-4e4c-4c08-995b-4f015c15d54b","resolution":{"observed_at":"2026-08-06T21:57:09.932628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:09.686638Z","title":"Harmonic experience: Tonal harmony from its natural origins to its modern expression","venue":null,"work_id":"82931eb2-d43d-4590-ae96-baa103cf70fc","year":1997},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:06.597284Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:76a4b6c8568d082f6b2c5261064b60c791be7deb29d143b725a82d3f42c59348","observation_id":"606c7818-3166-49b8-b16d-772e748a5a56","resolution":{"observed_at":"2026-08-06T21:57:09.751371Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-08-17T03:25:04.404839Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-06T21:57:06.742697Z","title":"Phi-3 technical report: A highly capable language model locally on your phone","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:06.742697Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:b179be974ae59e441d6054389600fc3c48d4dd68a2704b8824f0ec5bcb677764","observation_id":"4676f4b5-d193-46bb-9439-90bd09a41d67","resolution":{"observed_at":"2026-08-06T21:57:06.742697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:09.517446Z","title":"Trins: Towards multimodal language models that can read","venue":null,"work_id":"401b6195-d4ec-433d-8fb2-08b1ce8bc8c3","year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:06.908892Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:cc8587c05c161995d08f471d838a38850fa64c946ea56d103386443151e9032b","observation_id":"caa80afb-5dd9-4a46-8e4a-9b2ebfd4c561","resolution":{"observed_at":"2026-08-06T21:57:09.577261Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.19185","last_updated":"2024-07-27T05:53:37Z","snapshot_observed_at":"2026-08-16T13:30:36.830265Z","submitted_at":"2024-07-27T05:53:37Z","title":"LLaVA-Read: Enhancing Reading Ability of Multimodal Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.19185","snapshot_observed_at":"2026-08-06T21:57:07.053168Z","title":"Llava-read: Enhanc- ing reading ability of multimodal language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:07.053168Z"},"links":{"cited_paper":"/paper/2407.19185","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:b565fd2d70e58b1f7d97e9a46c236b70c3b7081cc0c61a8d50acce523e6caa95","observation_id":"97993c1c-9a6d-4154-9aff-ae7d5edad3c3","resolution":{"observed_at":"2026-08-06T21:57:07.053168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-20T11:47:17.477107Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-06T21:57:07.157500Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:07.157500Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:88e0907fb18bdf29eaba03bb46bc6ada5c0025de891b17d89cd05304c30d2e3d","observation_id":"4e90a859-a7cb-491c-bffa-8b713cd65997","resolution":{"observed_at":"2026-08-06T21:57:07.157500Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:09.335395Z","title":"Music information processing using the humdrum toolkit: Concepts, examples, and lessons","venue":null,"work_id":"d6e027d5-18cd-48aa-9879-45acd02a01f0","year":2002},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:07.301593Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:ea863bec4aa63e4126ec4437f2d36fe608dd42f6b019cea6b58da5dbe3b0f0df","observation_id":"c63f5167-f5c7-4aaa-978c-3c7e8c33e36c","resolution":{"observed_at":"2026-08-06T21:57:09.407200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14594","last_updated":"2024-08-26T19:26:50Z","snapshot_observed_at":"2026-08-16T13:23:42.025691Z","submitted_at":"2024-08-26T19:26:50Z","title":"MMR: Evaluating Reading Ability of Large Multimodal Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14594","snapshot_observed_at":"2026-08-06T21:57:07.432546Z","title":"MMR: Evaluating reading ability of large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:07.432546Z"},"links":{"cited_paper":"/paper/2408.14594","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:cc49d86c4db0411a91ffc43dab3e0d2eaa54a99629ea465ef9c4727e2c8deb59","observation_id":"1bdd0ad0-d2cc-4fb0-9f03-9aee76356e82","resolution":{"observed_at":"2026-08-06T21:57:07.432546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-14T20:13:52.872565Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-06T21:57:07.577643Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:07.577643Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:a413e8927c5c6068705b42993ac3bebe2c524b9c6bf32d639f4d41372b281e63","observation_id":"7e0f840c-504c-4ca6-b0f3-13ac051aa37e","resolution":{"observed_at":"2026-08-06T21:57:07.577643Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:07.694170Z","title":"Retrieval-augmented generation for knowledge- intensive nlp tasks","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:07.694170Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:78bb937eb3921a63cecc6ab3305b55a59904acd6fba7d3426539ca65e9eb1355","observation_id":"8b1be671-60c9-4af2-9463-b46312da531f","resolution":{"observed_at":"2026-08-06T21:57:07.694170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:09.171703Z","title":"Layoutgpt: Compositional visual planning and generation with large language models","venue":null,"work_id":"5c8a6d65-94d9-44d4-b7a3-7283ae5c2841","year":2023},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:07.851431Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:42c2ee6b360ff3a66be67f969ea320373bcd5a38d88daf603778e14217924160","observation_id":"ffefab5a-ed6e-485d-8181-41742bf42140","resolution":{"observed_at":"2026-08-06T21:57:09.227377Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12844","last_updated":"2024-10-09T19:51:38Z","snapshot_observed_at":"2026-08-19T16:48:50.845750Z","submitted_at":"2024-10-09T19:51:38Z","title":"TextLap: Customizing Language Models for Text-to-Layout Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12844","snapshot_observed_at":"2026-08-06T21:57:07.934980Z","title":"Text- lap: Customizing language models for text-to-layout planning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:07.934980Z"},"links":{"cited_paper":"/paper/2410.12844","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:5928ce4e8c007eb380991695983fe337cb951728f646a6b5c3986ec504cd8e28","observation_id":"a1c2e789-ffe0-480d-9692-972da2b8b98d","resolution":{"observed_at":"2026-08-06T21:57:07.934980Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-16T14:08:19.332089Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T21:57:08.059391Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:08.059391Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:75061f6f2f5f39bef1c7fed0720ea70ba3d2587a673d34482625b6bfbefa9078","observation_id":"29ef0fcd-2bc9-4672-916c-740cfec8368e","resolution":{"observed_at":"2026-08-06T21:57:08.059391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:57:09.007605Z","title":"Information not found","venue":null,"work_id":"543913b7-6330-4560-91c2-c9cadf317cd3","year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:08.122917Z"},"links":{"citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:4d9e8cb4cdd385527ba1dfb8adcc21ef6d0b4af57f19e16bc99094bd41ddbc3c","observation_id":"b87514ec-34f8-4519-91d0-5228c6da73f1","resolution":{"observed_at":"2026-08-06T21:57:09.066912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-13T02:11:32.989886Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":19,"verified_exact":2,"verified_fuzzy":28},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 5 inbound Pith citation observations for arXiv:2506.23009."}