{"as_of":"2026-08-10T13:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8e6d95d68c60d2e0908a987b3be13d87879c0f2b444156c4431cf2adb6ec0e9a","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:35:06.767136Z","state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:35:03.959317Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T21:35:06.839984Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"cited_work":{"arxiv_id":"2506.23873","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.23873","snapshot_observed_at":"2026-08-06T21:35:06.839984Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","venue":"cs.SD","work_id":"e3e24ce2-0f90-45b7-8c4a-152879b24ae1","year":2025},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:03.959317Z"},"links":{"cited_paper":"/paper/2506.23873","citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:8fb4d951a4b1969d3309a18b3aa6953554a0d97536126799cebe9bf23e196fe8","observation_id":"f8824715-575b-44c0-80e4-047e53cb353a","resolution":{"observed_at":"2026-08-06T21:35:06.846820Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.23873/citation-record","integrity":"/paper/2506.23873/integrity","json":"/paper/2506.23873/citation-record.json","paper":"/paper/2506.23873"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.726396Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","venue":null,"work_id":"58dc10f4-8406-46a8-8681-bd6036b0ce66","year":2025},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:03.892432Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:a0dd43ea4a7bef17f01e1251ec7bd6ea772b5f11009021e8f112f762e1f6e313","observation_id":"528fbdf2-b88e-4234-b372-c74b9fd65329","resolution":{"observed_at":"2026-08-06T21:35:07.730669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.711445Z","title":null,"venue":null,"work_id":"fd352f51-ccbb-472e-b59e-bb135242f2da","year":null},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.021261Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:927fd6d500ce9ac0eae049f1c2f5e11e9dc45da2615c521dec9c03c1a256ca79","observation_id":"cb29155c-c71a-4b9c-ba2a-12dbec1b318e","resolution":{"observed_at":"2026-08-06T21:35:07.716665Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.697696Z","title":"no chord","venue":null,"work_id":"4736bcc1-79d6-41a5-a694-bc16a8078d2c","year":null},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.056523Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:f32d56fbd237168519dfc4647f31a51e1d3e0894051d6c90a4dcfda39954ab35","observation_id":"863e3748-17fb-4e42-a69a-dac2464abb00","resolution":{"observed_at":"2026-08-06T21:35:07.702211Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.683817Z","title":"We also assess their contribution to global tasks","venue":null,"work_id":"52d2e020-812d-4bfd-b04b-e5f119a31892","year":null},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.127147Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:76cc622b7ac8f850edc632d57135afed43d20ce489273b2912862d32ea2ae90f","observation_id":"c9526bc9-2ade-4e72-8da3-6f1c7f2eeb58","resolution":{"observed_at":"2026-08-06T21:35:07.688335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.666964Z","title":"ViT-1D has 12 layers in total","venue":null,"work_id":"5a857e5d-5a0d-4d5c-9d11-c684e89f8bc1","year":null},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.179366Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:67b369ee6c9bc244fc135ba51ac5e1ac55022018a1bf9aa5dbee3e19e958c55e","observation_id":"7bd71412-e790-41c1-9ad5-fef69fe9df2e","resolution":{"observed_at":"2026-08-06T21:35:07.671704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.651589Z","title":",zT k ] at layers k = 3, 6, 9, 12 (same as Section 5, denoted z3 to z12), along with tokens from a randomly initialized ViT-1D model, de- noted zr","venue":null,"work_id":"b76f0e3f-9138-49ec-9b2f-a6fe7cfb2612","year":null},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.219993Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:10e46274640ac42f700768dc659cb875e0d360337d06a12ba365e61c1594c63e","observation_id":"6ce22dc1-2153-44d0-af30-1407ef227914","resolution":{"observed_at":"2026-08-06T21:35:07.656610Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.636800Z","title":"Applying NT-Xent loss only to the class token in a lightweight ViT-1D surprisingly enables sequence tokens to handle local tasks while con- tributing to global ones","venue":null,"work_id":"43e5f0f6-5c8c-4e1d-9861-45d1aa53f45b","year":null},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.298659Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:9f33354bd91898e57111a845d222315b25bb1f6a547a6591cc7e40e1c0abf8dc","observation_id":"f59d7daf-a86b-4800-aefc-1a5c815c5819","resolution":{"observed_at":"2026-08-06T21:35:07.641602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.529154Z","title":"Contrastive learn- ing of musical representations,","venue":null,"work_id":"44bf145c-5ef2-4833-9c56-a49328189a44","year":2021},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.753468Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:206e4ec90e83b9349c1cd082168baf8c88d91ab3b72652ae47a7c6c88272071a","observation_id":"d3466fb9-5df9-4222-a525-8f8ec989325f","resolution":{"observed_at":"2026-08-06T21:35:07.534811Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"cited_work":{"arxiv_id":"2506.23873","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.23873","snapshot_observed_at":"2026-08-06T21:35:06.839984Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","venue":"cs.SD","work_id":"e3e24ce2-0f90-45b7-8c4a-152879b24ae1","year":2025},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:03.959317Z"},"links":{"cited_paper":"/paper/2506.23873","citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:8fb4d951a4b1969d3309a18b3aa6953554a0d97536126799cebe9bf23e196fe8","observation_id":"f8824715-575b-44c0-80e4-047e53cb353a","resolution":{"observed_at":"2026-08-06T21:35:06.846820Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.621392Z","title":"Convolutional operators in the time- frequency domain,","venue":null,"work_id":"6c4b751f-7352-4f2f-a6cb-49aede80fb82","year":2017},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.378357Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:2ac03a180a9af3c8012d2dfc2c24165348e7c90f3d2f37ae7c398111594935b3","observation_id":"c7500886-066b-4b50-b661-9fecfb08d8c3","resolution":{"observed_at":"2026-08-06T21:35:07.626355Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.605632Z","title":"Pesto: Pitch estimation with self-supervised transposition-equivariant objective,","venue":null,"work_id":"1db96a61-ce45-4a9a-b5d5-c77afaf2458c","year":2023},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.434753Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:42ca1e88c17deb012b5ccb98c7de579397395ed27e2194a2d2425dbc7eb877d9","observation_id":"ee3b7738-b9c7-4256-8402-d824ef6a9fbe","resolution":{"observed_at":"2026-08-06T21:35:07.610762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.589119Z","title":"Equivariant self-supervision for musical tempo estimation,","venue":null,"work_id":"3ff9ebc5-4990-4e7e-b150-6e7eaa188fe5","year":2022},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.497752Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:02e674b307d58c1becd60d22bc83eb1c2b1fd0e25acacdd9039260b571bcee1f","observation_id":"1fc77fcb-fae8-4e59-acd8-47eea56e370a","resolution":{"observed_at":"2026-08-06T21:35:07.593569Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.574174Z","title":"STONE: Self- supervised tonality estimator,","venue":null,"work_id":"3b780aec-a643-499a-83be-4464ebfa7584","year":2024},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.566969Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:cb28c48a5283c7e643fed12953b581b7092c125cff48998334c47271343016dd","observation_id":"d38845ea-aff5-4295-8489-914c11e7db75","resolution":{"observed_at":"2026-08-06T21:35:07.579191Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.559957Z","title":"S-key: Self-supervised learning of major and minor keys from audio,","venue":null,"work_id":"718bef61-9019-4542-9796-650bdf231a44","year":2025},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.615860Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:c8f77b161015d5205f51fc9f4a20b25fc671eeed9cb0060db10baf20059d852e","observation_id":"a4b40a51-bb0e-4bf4-9f7c-d7feafde0300","resolution":{"observed_at":"2026-08-06T21:35:07.564180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.546118Z","title":"Data cleansing with contrastive learning for vocal note event annotations,","venue":null,"work_id":"4c68add2-36b1-4070-8e20-c5cba6a32977","year":2020},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.661655Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:0e84c71a0c013f668e0f8c8fba3a332818dc2d996b2772439e62f6706331b2e7","observation_id":"b07e0440-454d-402f-bc93-2a599556c5cb","resolution":{"observed_at":"2026-08-06T21:35:07.550434Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14340","last_updated":"2024-09-03T14:53:34Z","snapshot_observed_at":"2026-07-06T19:06:01.833508Z","submitted_at":"2024-08-26T15:13:14Z","title":"Foundation Models for Music: A Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14340","snapshot_observed_at":"2026-08-06T21:35:04.720630Z","title":"Foundation models for music: A survey,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.720630Z"},"links":{"cited_paper":"/paper/2408.14340","citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:1ee16b896d59d68d709115022bbca62941e24a638dd41a172d23bb531bd98aa5","observation_id":"865fbb87-2a6a-4007-86df-e97febd206a5","resolution":{"observed_at":"2026-08-06T21:35:04.720630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.515514Z","title":"Supervised and un- supervised learning of audio representations for music understanding,","venue":null,"work_id":"4def9356-f950-4f29-b4a4-1a22537c4a75","year":2022},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.825973Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:565b462933c71f88beb4131220546969bfc814147b738d55273d212ff4ace58c","observation_id":"bdbe5d22-1502-4b3b-a90d-4e42b4422e2a","resolution":{"observed_at":"2026-08-06T21:35:07.519820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.498357Z","title":"A simple framework for contrastive learning of visual representations,","venue":null,"work_id":"b59070ab-9bda-4eab-90d3-aed292149cd7","year":2020},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:04.870126Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:f5c563b9ed8c61cbe621e0cb948af28254ad53f7e8791c03265e0dd79586c93d","observation_id":"58c03637-db8d-4e40-8338-de22e327f3aa","resolution":{"observed_at":"2026-08-06T21:35:07.503301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.481464Z","title":"Exploring simple siamese rep- resentation learning,","venue":null,"work_id":"6d0d131f-832a-45c3-9c1e-4bb10b9d6044","year":2021},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:05.003557Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:ef998a9fe3024dfe65b3401f74463e5ad9d4e65c15f22511ebc7728db8ba55bf","observation_id":"4e614b17-e1c1-4134-9d65-12dae3c6600c","resolution":{"observed_at":"2026-08-06T21:35:07.486063Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.466316Z","title":"S3t: Self-supervised pre-training with swin transformer for music classification,","venue":null,"work_id":"66df6c79-059e-4d29-a5a5-31034a8110ed","year":2022},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:05.126339Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:52d10343339ff0a67d1b9c270f02696653906f457958dd40ec73f7b2de4e923f","observation_id":"458e0e72-5dee-4e42-9a35-276709968ecb","resolution":{"observed_at":"2026-08-06T21:35:07.470434Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.452370Z","title":"Multi- source contrastive learning from musical audio,","venue":null,"work_id":"89244c71-629d-42f6-8d5f-8faf9cbcc76f","year":2023},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:05.208870Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:47ae481eab591076850b795093aabd67f933a766c1db3b60bf8979a8fe3386f8","observation_id":"4d988db2-b634-4d48-ad42-bd4fba13f267","resolution":{"observed_at":"2026-08-06T21:35:07.456811Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.438475Z","title":"On the effect of data-augmentation on local embedding properties in the contrastive learning of music audio representations,","venue":null,"work_id":"9fe4e449-689e-4f56-8211-593ab8177c15","year":2024},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:05.289915Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:6fc8110f6d0478a175c6b244e2c7de64adf89f7e99e1221977fdd684b1057070","observation_id":"a3bbf4ac-2314-45cb-8010-ec63bc69de56","resolution":{"observed_at":"2026-08-06T21:35:07.442948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.424777Z","title":"Towards proper contrastive self-supervised learning strategies for mu- sic audio representation,","venue":null,"work_id":"1dac03d5-2b83-4d3c-b1e7-9cb3a50b4c41","year":2022},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:05.396444Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:78d859e69ed618bf3a78b11152d46f2a6f6c7c6b4beb7a5e79fc140adc61ded9","observation_id":"96188798-a9b3-4f25-ad9b-c73e1b069693","resolution":{"observed_at":"2026-08-06T21:35:07.429137Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.00341","last_updated":"2020-04-30T09:02:45Z","snapshot_observed_at":"2026-08-06T03:57:57.420510Z","submitted_at":"2020-04-30T09:02:45Z","title":"Jukebox: A Generative Model for Music","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.00341","snapshot_observed_at":"2026-08-06T21:35:05.476840Z","title":"Jukebox: A generative model for music,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:05.476840Z"},"links":{"cited_paper":"/paper/2005.00341","citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:722e45f45000df35d3ffd8d177b171d8aa4f94328dd29c6a0e4806f8bff242d4","observation_id":"fd279e44-5c74-4ff4-a46d-9d74ec13a658","resolution":{"observed_at":"2026-08-06T21:35:05.476840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.409266Z","title":"Music2latent: Consistency autoencoders for latent audio compres- sion,","venue":null,"work_id":"8003c1df-693b-4a92-b4a6-10b1285d288e","year":2024},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:05.527201Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:3f342ef4a89ee8c01f509c1f2e6c5b7cb113b30bcf62a9685334168e340a5ccb","observation_id":"2aaf93c1-9bbb-4bc0-9a6e-78093de39268","resolution":{"observed_at":"2026-08-06T21:35:07.414397Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.395177Z","title":"Mert: Acoustic music understand- ing model with large-scale self-supervised training,","venue":null,"work_id":"2adb798a-ab7e-4be2-8def-d03f5f3374b7","year":2023},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:05.632281Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:5e5ae09535f54fe97ca40812a018ef649dbcc0fd294d2ba25fc27d4df38d3251","observation_id":"17ce145e-450e-4671-8857-7628a823c927","resolution":{"observed_at":"2026-08-06T21:35:07.399602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.380051Z","title":"Hu- bert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":"cf2ca80a-eefa-43f0-af7f-19a1cff71657","year":2021},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:05.764724Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:008351eb342a0948f6db935ad5cf2be181e5cab272f3e07c8c4ff8260b87f7b4","observation_id":"bf08a89e-44f0-4548-adcd-7ea92fc5e838","resolution":{"observed_at":"2026-08-06T21:35:07.384924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.364174Z","title":"Masked Modeling Duo: Towards a Universal Audio Pre-training Framework,","venue":null,"work_id":"754a257c-f7a2-4a36-9124-211f6f3647a9","year":2024},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:05.929410Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:134129d2c292e6d63dc2585d45ba1c30bf63fe335dbb4272becb13c8558540c7","observation_id":"a36c6b53-2ee3-4af9-ac30-dd0535a7b306","resolution":{"observed_at":"2026-08-06T21:35:07.369013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.348627Z","title":"A foundation model for music informatics,","venue":null,"work_id":"f95bbe5e-8071-47a4-9d68-28491b8c4a11","year":2024},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.059702Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:588ef0ccd628498e0250f4f96db57f1b6024a0d4d6488eeae1c534470e794d41","observation_id":"9ed29d95-722a-4f58-8a8c-b338ce9d31e7","resolution":{"observed_at":"2026-08-06T21:35:07.354113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.334780Z","title":"Towards learning universal audio repre- sentations,","venue":null,"work_id":"8402b38a-8689-4d67-afcd-9b47d938a0fc","year":2022},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.198367Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:53497bb83d4f865e74f0a9f1b334dcf9fce08a4857dd7a2c5c24c791cc5bba93","observation_id":"55d2706a-a363-4f84-8e75-fca37ed507c6","resolution":{"observed_at":"2026-08-06T21:35:07.339180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.321132Z","title":"Efficient training of audio transformers with patchout,","venue":null,"work_id":"e7fc5ed9-242f-4da2-b3d8-c2eba1ce7faf","year":2022},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.315056Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:7e5e2b560da6afe2859a344599a4926810f05befb9c732e965bc2625c2ee9094","observation_id":"899d05f0-b107-4a45-8970-142757f7b261","resolution":{"observed_at":"2026-08-06T21:35:07.325628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.307103Z","title":"AST: Audio Spectrogram Transformer,","venue":null,"work_id":"d6e6660c-2eba-485b-ac66-98e57b23021b","year":2021},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.472452Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:44d859876bfc3ed1c37b19d5db8cb20aaa57c00730537e9b6cf0694e2a921e41","observation_id":"f5c98dbb-9457-4fed-803d-2f660091463a","resolution":{"observed_at":"2026-08-06T21:35:07.311361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.292295Z","title":"Contrastive audio-language learning for music,","venue":null,"work_id":"9ae34b24-0c91-4ac0-adbe-f7cdc28b668c","year":2022},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.639266Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:f5c1e2ee76d2754db77bcf0639ae757db1c3bfa82d9a6e73d658c850ddb44953","observation_id":"6d486a54-0999-4ea3-8da7-b63a8d4957d1","resolution":{"observed_at":"2026-08-06T21:35:07.297146Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.278018Z","title":"MuLan: A joint embedding of music audio and natural language,","venue":null,"work_id":"c1369bf8-4398-4dd7-9ad0-3090e4f56c57","year":2022},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.674529Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:20ddb8a43109adf576037c2f7908609ae5974e9ca62c703c06eab378b06e0ed1","observation_id":"074b5f9a-a406-4fbf-9f62-3a7b9d51397f","resolution":{"observed_at":"2026-08-06T21:35:07.282242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.264540Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale,","venue":null,"work_id":"c1479f83-9f57-4b52-8ff9-0b66e7a7bf36","year":2021},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.679406Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:bfb6da5b30cb7ca9a271d9010f968fae4e3dc46eddfba0a255374b8458c05ff3","observation_id":"98d301f3-89f8-4c4a-8df1-dc3832eb69e6","resolution":{"observed_at":"2026-08-06T21:35:07.268884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.250859Z","title":"Emerging properties in self-supervised vision transformers,","venue":null,"work_id":"a2fa424f-d7c6-4f73-b1c4-5819876edf0c","year":2021},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.683674Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:f94730aca38d4e40d2494d2dfdc89620f967d3bf976c2e65c18e07fd67a37df7","observation_id":"0ddd29b2-f065-40fb-8084-8d812f25c609","resolution":{"observed_at":"2026-08-06T21:35:07.255325Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.236862Z","title":"Dinov2: Learning robust visual features without su- pervision,","venue":null,"work_id":"f62f46a1-cd6d-4d57-aa3c-fba38f080099","year":2023},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.688939Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:52473ee13e88f3e0acc47e18aaa76af062c66766d57dd73bd8024cacf8a9e4ce","observation_id":"d749dca6-d96e-4f2a-ab1b-82d70efb1fa3","resolution":{"observed_at":"2026-08-06T21:35:07.241574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.222270Z","title":"An experimental comparison of multi-view self-supervised methods for music tagging,","venue":null,"work_id":"08ec03f5-1854-4615-a9b8-373d3d7424bf","year":2024},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.693189Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:066ef8dac5f1add0a3c362451e2315d85d7f55fcfa9ce098134df6f6cd96bf61","observation_id":"226d7449-df82-40ae-a6f2-03f66a9dc5c3","resolution":{"observed_at":"2026-08-06T21:35:07.227042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.207662Z","title":"Evaluation of algorithms using games: The case of music tagging,","venue":null,"work_id":"e3ffd1bf-dd31-4e82-b03b-328ced083129","year":2009},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.697437Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:95d99d650aa2d13d22d59440f0a893fa80af3b13ad362d3688a9ff069b7fec60","observation_id":"13f4b01c-0950-48c2-a97d-ffc126b7b848","resolution":{"observed_at":"2026-08-06T21:35:07.212178Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.192631Z","title":"Sample-level deep convolutional neural networks for music auto- tagging using raw waveforms,","venue":null,"work_id":"56e73e5f-3597-4cb9-b6f5-c3c357a0e8bf","year":2017},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.702513Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:e876ff2ebf1fa5b86069b6d19223ad2ecc17ef2e571c6ea927ae4f2450134345","observation_id":"e28d92d8-1aab-41b5-ae3a-56c8a7637f0d","resolution":{"observed_at":"2026-08-06T21:35:07.197354Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.178282Z","title":"Fmak: A dataset of key and mode annotations for the free music archive– extended abstract,","venue":null,"work_id":"0f1a3de5-b6c0-4956-8d7f-91881ac0cccc","year":2023},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.707049Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:525eb569ca1899979955139ec7d825bedd1179e8307205f26d4671945cf77a73","observation_id":"0bfdffd7-e98b-44b1-965b-5f578e7b2b94","resolution":{"observed_at":"2026-08-06T21:35:07.182847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.164281Z","title":"Fma: A dataset for music analysis,","venue":null,"work_id":"24e684a7-03bc-4755-8050-3efc4e7c32d2","year":2017},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.711659Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:99c494c9aa38cc46f05469fa0a7044704d8d229b271f3a13013663b0c7828c34","observation_id":"fa88e5c1-f5a9-4383-a79c-b8aab4f90d48","resolution":{"observed_at":"2026-08-06T21:35:07.168855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.150817Z","title":"Two datasets for tempo estimation and key detection in electronic dance music annotated from user corrections,","venue":null,"work_id":"5d146e79-60ad-41af-be3d-e9d3a9e34c34","year":2015},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.715628Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:6f673dc6e9110207018e9b02d80f465fa7e9eebd03a1e1b39102edc74371d738","observation_id":"4261a79d-38a1-49c8-963a-6dcde246119f","resolution":{"observed_at":"2026-08-06T21:35:07.155065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.137975Z","title":"Mir_eval: A transparent implementation of common mir metrics","venue":null,"work_id":"5be771fd-2509-490b-934d-45d32335dc8a","year":2014},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.719555Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:78694ffd388d505d84b4ce4ecf411a5f9b8aea0eb68525dad1d58dfae72256c4","observation_id":"e23889d3-0cbb-4bd8-af08-b83929fd044d","resolution":{"observed_at":"2026-08-06T21:35:07.141911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.124094Z","title":"A review of rhythm de- scription systems,","venue":null,"work_id":"81df7c1d-fe9e-438a-810a-abba9fec0e1a","year":2004},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.724174Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:71f2f621ede09a83dfa4ac877893f7ed533168af46e66050881900afe816653d","observation_id":"2c3cc49e-13ae-4d75-b100-b650bc73cc34","resolution":{"observed_at":"2026-08-06T21:35:07.128388Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.109495Z","title":"Gtzan- rhythm: Extending the gtzan test-set with beat, down- beat and swing annotations,","venue":null,"work_id":"d163de2c-6fc9-4581-87c0-e58d7a2959ef","year":2015},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.729003Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:cf8f88a74ebb191878427597c9ce5d5d4eb6b5777d00b8e9f68b29ca6c7dff84","observation_id":"e8384560-86f0-475e-a6dc-a925eb01e127","resolution":{"observed_at":"2026-08-06T21:35:07.114137Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.095941Z","title":"An efficient state- space model for joint tempo and meter tracking","venue":null,"work_id":"caa1a2cf-2992-4948-aa4f-966f9df54e6b","year":2015},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.733502Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:c726964221f4e0c952ce7c3568d08b33f36fc3432cff35afee008fb768622c37","observation_id":"29e5a141-58f7-462c-ae10-9a1b598ad9d0","resolution":{"observed_at":"2026-08-06T21:35:07.100350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:07.081490Z","title":"Schubert win- terreise dataset: A multimodal scenario for music anal- ysis,","venue":null,"work_id":"7de65542-0184-49cf-8caa-0b3aa76e0472","year":2021},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.738154Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:88b6ff2deb7192e533ce447a1d09274595ea3b53e97713651e6298e0a80913ad","observation_id":"3b5e2a55-18bf-417a-9cae-bc64d6d9fe20","resolution":{"observed_at":"2026-08-06T21:35:07.086494Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:06.947348Z","title":"Rwc music database: Popular, classical, and jazz music databases,","venue":null,"work_id":"3bb76631-d976-4eee-a64d-f41d3f5a10ba","year":2002},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.742072Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:7829da7fe0f2c5782c8e9dce902bc6586b1703666c3daa684e96ae4d6ab1a9d6","observation_id":"1caf7440-4e60-42e1-9807-91284e18a930","resolution":{"observed_at":"2026-08-06T21:35:06.951477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:06.932520Z","title":"Attention is all you need,","venue":null,"work_id":"a347080b-df2c-4b0f-b082-33ff736164fd","year":2017},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.746057Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:0e3e8abf877908ad68d1328ecc632dc3b5cb06ce49d2c3f8a099ea76482e0642","observation_id":"4e433577-08ba-47a5-a0e9-024df2af5a6b","resolution":{"observed_at":"2026-08-06T21:35:06.937407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:06.915667Z","title":"Sbert-wk: A sentence em- bedding method by dissecting bert-based word mod- els,","venue":null,"work_id":"088a8fad-c1f9-4fbe-98e8-27f85a4ad0f7","year":2020},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.750056Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:b7fc6dba21b70ab9bfa150b33e78f58bed2404a7a71de8828e24234fb2f00687","observation_id":"c959286e-801c-4fea-b083-faeeb55374d1","resolution":{"observed_at":"2026-08-06T21:35:06.921383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:06.901726Z","title":"Multipitch esti- mation of piano sounds using a new probabilistic spec- tral smoothness principle,","venue":null,"work_id":"9b09329d-99d4-45f3-b357-1be012bad4bf","year":2009},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.754169Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:c8993e62a8351307b911f8c5be1f4298efa7cb4fa723f71ee45a2cb5e0b982b0","observation_id":"0efdfb1f-c557-4300-9d88-530e218306df","resolution":{"observed_at":"2026-08-06T21:35:06.905846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:06.887680Z","title":"SciPy 1.0: Fundamental Algorithms for Scientific Computing in Python,","venue":null,"work_id":"09e074c4-73de-4001-8c36-14b996498a46","year":2020},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.758358Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:085beafcb5eaeeaf0e82b6de3d6a1eaa04f64a2ba6aceae971fed9866dd44211","observation_id":"34d065b2-aead-4b5b-a7e5-bb482dc064bf","resolution":{"observed_at":"2026-08-06T21:35:06.892124Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:06.872131Z","title":"How many layers and why? An analysis of the model depth in transformers,","venue":null,"work_id":"14349c26-0e18-4f19-af04-c4642a35acf7","year":2021},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.762767Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:0aa16b45590c112c0f8d1815cf145f345b9e37d0371b23f6f2fe1932c2f1a559","observation_id":"8f7bef19-b1bc-43b1-9700-72719f01615b","resolution":{"observed_at":"2026-08-06T21:35:06.876785Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:35:06.857643Z","title":"What does bert look at? an analysis of bert’s attention,","venue":null,"work_id":"c1fceafd-e32b-4163-b6f0-6746a2390525","year":2019},"citing_paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T21:35:06.767136Z"},"links":{"citing_paper":"/paper/2506.23873"},"observation_digest":"sha256:80e9cd8ff6e14dbdec98722349a192e5d46cbfe99f371764343c99ee46012b5f","observation_id":"29c93086-8781-4103-8290-204c23bc153c","resolution":{"observed_at":"2026-08-06T21:35:06.862038Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.23873","last_updated":"2025-06-30T14:04:59Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-06T21:27:13.599776Z","submitted_at":"2025-06-30T14:04:59Z","title":"Emergent musical properties of a transformer under contrastive self-supervised learning"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":3,"verified_exact":0,"verified_fuzzy":51},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 1 inbound Pith citation observation for arXiv:2506.23873."}