{"as_of":"2026-08-14T18:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2392a3f6ee8cf27aa6ed3c9f77d20944304a59206818ca990ab0bc6a2e6258d1","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T12:21:15.520982Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T11:13:58.950328Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-29T18:23:51.303795Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"cited_work":{"arxiv_id":"2507.21917","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.21917","snapshot_observed_at":"2026-06-29T18:23:51.303795Z","title":"& Castellano, G","venue":null,"work_id":"bff5d480-9b89-40b1-9426-a1b05a46e3a1","year":2025},"citing_paper":{"arxiv_id":"2603.18472","last_updated":"2026-04-09T02:35:56Z","snapshot_observed_at":"2026-07-06T22:49:37.944352Z","submitted_at":"2026-03-19T04:08:20Z","title":"Cognitive Mismatch in Multimodal Large Language Models for Discrete Symbol Understanding","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-15T09:11:31.870441Z"},"links":{"cited_paper":"/paper/2507.21917","citing_paper":"/paper/2603.18472"},"observation_digest":"sha256:0309728e0f0845e1c96c122a16f1524ac23293f9ef9d517d6087976a40e57669","observation_id":"40f2e3ed-7984-40f9-bb3f-3843d57451db","resolution":{"observed_at":"2026-05-15T09:15:21.084154Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"cited_work":{"arxiv_id":"2507.21917","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.21917","snapshot_observed_at":"2026-06-29T18:23:51.303795Z","title":"& Castellano, G","venue":null,"work_id":"bff5d480-9b89-40b1-9426-a1b05a46e3a1","year":2025},"citing_paper":{"arxiv_id":"2604.07338","last_updated":"2026-04-08T17:53:26Z","snapshot_observed_at":"2026-08-11T12:41:43.815777Z","submitted_at":"2026-04-08T17:53:26Z","title":"Appear2Meaning: A Cross-Cultural Benchmark for Structured Cultural Metadata Inference from Images","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T17:49:25.227568Z"},"links":{"cited_paper":"/paper/2507.21917","citing_paper":"/paper/2604.07338"},"observation_digest":"sha256:2d1d7dafea4b25ea26bde819450776093afe2e905105574f20a9bcc074ec2608","observation_id":"e6393261-fd1e-4261-a288-b4a7d816a19c","resolution":{"observed_at":"2026-05-11T06:05:55.467704Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"cited_work":{"arxiv_id":"2507.21917","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.21917","snapshot_observed_at":"2026-06-29T18:23:51.303795Z","title":"& Castellano, G","venue":null,"work_id":"bff5d480-9b89-40b1-9426-a1b05a46e3a1","year":2025},"citing_paper":{"arxiv_id":"2606.27947","last_updated":"2026-06-26T10:42:43Z","snapshot_observed_at":"2026-08-08T10:08:09.608301Z","submitted_at":"2026-06-26T10:42:43Z","title":"Understanding How MLLMs Describe Artworks Using Token Activation Maps","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T05:06:35.085802Z"},"links":{"cited_paper":"/paper/2507.21917","citing_paper":"/paper/2606.27947"},"observation_digest":"sha256:0e59d78fbdc126a7e19b5291de4a6ce62215a43cc9e48a0e6232da02162f4c10","observation_id":"710e3f64-12c0-4227-a319-519d0174c297","resolution":{"observed_at":"2026-06-29T18:23:51.305082Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.21917","snapshot_observed_at":"2026-08-06T11:13:58.950328Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05026","last_updated":"2026-08-05T16:30:12Z","snapshot_observed_at":"2026-08-14T01:13:04.829928Z","submitted_at":"2026-08-05T16:30:12Z","title":"ArtAnno: Annotating Implicit Semantics in Artworks through LLM Agent-Driven Bidirectional Human-AI Augmentation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T11:13:58.950328Z"},"links":{"cited_paper":"/paper/2507.21917","citing_paper":"/paper/2608.05026"},"observation_digest":"sha256:c595a885d993aef0aa952a43b04080a9db3d9044cd3cd0e969adf78f10a321a0","observation_id":"6879c8b8-a8b5-4248-bb49-3e95baadafc6","resolution":{"observed_at":"2026-08-06T11:13:58.950328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.21917/citation-record","integrity":"/paper/2507.21917/integrity","json":"/paper/2507.21917/citation-record.json","paper":"/paper/2507.21917"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.943480Z","title":null,"venue":null,"work_id":"8fae7dc8-10b7-4882-a3d7-b04e1d0f773d","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.376287Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:b4e7fd3368cd1d2010e819c4dd8268af2ffd81505f9aef83f0e60ba99e714286","observation_id":"e58c71a6-42b4-4ffa-b0b1-921d762e9479","resolution":{"observed_at":"2026-08-06T12:21:15.946480Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.934719Z","title":"Leveraging knowledge graphs and deep learning for automatic art analysis,","venue":null,"work_id":"9f2bd266-b74f-4be8-b407-a561ef520c77","year":2022},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.379980Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:32a3e5587d497076d2f663fe5332ab79933922787a0c1a4952c6770257f8979f","observation_id":"bb92f3e5-ee8b-4339-9da7-04b63b97b603","resolution":{"observed_at":"2026-08-06T12:21:15.937813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.926394Z","title":"GraphCLIP: Image-graph contrastive learning for multimodal artwork classification,","venue":null,"work_id":"cec8c821-14b9-4b26-80b4-238056c47074","year":2025},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.383516Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:cd3036797181b1b5147d6faea4943c5eaefe37aa15f06a75ac27dc9dd6662dec","observation_id":"85581239-a6e7-49b6-b7fe-088a5ce8c4e9","resolution":{"observed_at":"2026-08-06T12:21:15.929473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.918061Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":"4f142f88-c7f2-40c0-9de3-58f2ebcbce50","year":2021},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.386819Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:e6b3e141e09f090f26eed752f5ef8480a7a6e1820c28d17e718e51df6883bb62","observation_id":"09f67382-5b4e-41f5-b76b-b8608701fff6","resolution":{"observed_at":"2026-08-06T12:21:15.921080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T12:21:15.389939Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.389939Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:1b6918506dc11cb9169632fac3bd8be0e4a62eac861dc42dddedad6a01b6f927","observation_id":"89069abc-1406-4f9d-9b92-372b40e2c1cf","resolution":{"observed_at":"2026-08-06T12:21:15.389939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07490","last_updated":"2024-04-04T18:55:18Z","snapshot_observed_at":"2026-08-13T11:43:45.191377Z","submitted_at":"2023-05-12T14:04:30Z","title":"ArtGPT-4: Towards Artistic-understanding Large Vision-Language Models with Enhanced Adapter","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07490","snapshot_observed_at":"2026-08-06T12:21:15.393243Z","title":"ArtGPT-4: Towards artistic-understanding large vision-language models with enhanced adapter,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.393243Z"},"links":{"cited_paper":"/paper/2305.07490","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:f7540752dd73f628f9ce4d79b9a92a6143f40d6bdf12f1c99183072f98ecb5ad","observation_id":"82ddbe2f-079b-4080-8e8e-f674a9773db3","resolution":{"observed_at":"2026-08-06T12:21:15.393243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.908814Z","title":"Gallerygpt: Analyzing paintings with large multimodal models,","venue":null,"work_id":"fd9643bc-1b2e-4cdb-932b-181e4cec7cdf","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.396782Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:718762d8eaf1342f62ad08f3666e6cd3d8a690cbc136d8ed0e64ef3db1deafd8","observation_id":"303579f4-8f4d-4066-ae43-bd80e62d7ad1","resolution":{"observed_at":"2026-08-06T12:21:15.912785Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.10921","last_updated":"2024-09-17T06:39:18Z","snapshot_observed_at":"2026-08-13T16:12:29.629157Z","submitted_at":"2024-09-17T06:39:18Z","title":"KALE: An Artwork Image Captioning System Augmented with Heterogeneous Graph","version":1},"cited_work":{"arxiv_id":"2409.10921","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.10921","snapshot_observed_at":"2026-08-06T12:21:15.592463Z","title":"KALE: An Artwork Image Captioning System Augmented with Heterogeneous Graph","venue":"cs.CV","work_id":"8d13328a-731c-4703-97b7-a2e5af3e011f","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.399409Z"},"links":{"cited_paper":"/paper/2409.10921","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:e34c89c44926768c05b2ff8a12b156250fa51d84f0af82de92820b22232d1368","observation_id":"96df4660-a14e-427b-b6f7-17c4164ac575","resolution":{"observed_at":"2026-08-06T12:21:15.597754Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.900491Z","title":"Colbert: Efficient and effective passage search via contextualized late interaction over bert,","venue":null,"work_id":"ac422665-3445-471f-9291-82983ed65695","year":2020},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.402733Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:69ebf8ced514120ac86b31ec3bcf7fd724524201dc922481cf78b350528729b9","observation_id":"d17427ac-7cd6-4fd3-a2c1-15891a390e1a","resolution":{"observed_at":"2026-08-06T12:21:15.903528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.892289Z","title":"Artpedia: A new visual-semantic dataset with visual and contextual sentences in the artistic domain,","venue":null,"work_id":"9ba6b050-6558-42f0-aca9-68f2ab7f0bda","year":2019},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.405602Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:367c18cbcd3ab95f44aaf06029aee22c81a692e1d17e724ecfe80e3c2de7a9ef","observation_id":"5218236b-b32b-4c5f-ae03-183c3e8f9dfb","resolution":{"observed_at":"2026-08-06T12:21:15.895336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.883078Z","title":"Deep learning approaches to pattern extraction and recognition in paintings and drawings: An overview,","venue":null,"work_id":"1014c4c1-37fd-498d-ac0f-ce91244fd427","year":2021},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.408392Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:90507b2925a4e9e7d871cf0a7c9f299ec550bdb259fcc59abbe2e039d679cbea","observation_id":"cb8a1aa8-1494-4cab-973d-395f0f91ca35","resolution":{"observed_at":"2026-08-06T12:21:15.886284Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.874809Z","title":"Machine learning for cultural heritage: A survey,","venue":null,"work_id":"e7521fc7-b3e0-45e9-8d8e-b1451d21f803","year":2020},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.411512Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:81173d620dc5b2aadfa3d5705c105ac676f333c5413446003c55a222ce6af23c","observation_id":"bf5868df-3321-4f3c-991a-9a2cd3b0770f","resolution":{"observed_at":"2026-08-06T12:21:15.877773Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.866524Z","title":"Fine-tuning convolutional neural networks for fine art classification,","venue":null,"work_id":"a008f925-5a42-4198-aa01-99ff49805f3e","year":2018},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.415129Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:0c163315df3771b94528fccd4ca753605a223b1161d90f290880e4833fc0381c","observation_id":"fd1a1c4a-be3f-4538-8249-2b852a36360e","resolution":{"observed_at":"2026-08-06T12:21:15.869536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1311.3715","last_updated":"2014-07-23T07:56:20Z","snapshot_observed_at":"2026-07-06T03:28:14.741672Z","submitted_at":"2013-11-15T03:37:50Z","title":"Recognizing Image Style","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1311.3715","snapshot_observed_at":"2026-08-06T12:21:15.417940Z","title":"Recognizing image style,","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.417940Z"},"links":{"cited_paper":"/paper/1311.3715","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:701d2fa3605f97f3b7f1cc11a7a16c0c3f6d308572872fd5e24d48c8035e8d96","observation_id":"92586de4-cc32-4b0c-957d-e42eef8eea45","resolution":{"observed_at":"2026-08-06T12:21:15.417940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1505.00855","last_updated":"2015-05-05T01:25:26Z","snapshot_observed_at":"2026-07-06T04:16:54.272660Z","submitted_at":"2015-05-05T01:25:26Z","title":"Large-scale Classification of Fine-Art Paintings: Learning The Right Metric on The Right Feature","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1505.00855","snapshot_observed_at":"2026-08-06T12:21:15.421188Z","title":"Large-scale classification of fine-art paintings: Learning the right metric on the right feature,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.421188Z"},"links":{"cited_paper":"/paper/1505.00855","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:ad260de8966fefade51bc849e9925874d4e7ef10f3fd0e4b12c13c06e7e3640a","observation_id":"f80128db-6fba-4cf4-a641-c745b61c6bf9","resolution":{"observed_at":"2026-08-06T12:21:15.421188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.858112Z","title":"Toward Discovery of the Artist’s Style: Learning to recognize artists by their artworks,","venue":null,"work_id":"2eea0b59-2db5-49fb-9d2c-5d0ee61731d3","year":2015},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.424471Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:c26cd124cba23a58345ee7cf8d9ef66f5b8d8945dc86b9bbdc4faaefb0eb8991","observation_id":"1ff222b3-d6e5-4844-9d06-8e57cf17d884","resolution":{"observed_at":"2026-08-06T12:21:15.861235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.849928Z","title":"A deep learning approach to clustering visual arts,","venue":null,"work_id":"8804c547-c91b-430d-b97e-e4ec6c721e66","year":2022},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.428015Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:a7ab0c3502c926da71a03fc0cdb984b32c646049f2127cc86bbae403d48e4204","observation_id":"1f9b41be-672e-4017-8e92-b83db479d8b6","resolution":{"observed_at":"2026-08-06T12:21:15.852876Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.841627Z","title":"Toward automated discovery of artistic influence,","venue":null,"work_id":"d8a4f743-3e7a-4539-b77c-0546225b4b30","year":2016},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.431094Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:2f6652bc78ae28bf01af50320aab39ff494893b76791d723d031f16994ebdab6","observation_id":"f4e985b6-a115-4f2d-8495-4c3b7d2d6ce4","resolution":{"observed_at":"2026-08-06T12:21:15.844686Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.832955Z","title":"Wasielewski, Computational formalism: Art history and machine learning","venue":null,"work_id":"0ad475c6-b015-4288-8e2b-4c962f6263ee","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.433853Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:e2a86fb227e17c79c159adff048340b83e5857db5659e2886ddb4082f5f4aaf6","observation_id":"21280f40-d304-4669-b5fa-44b502e038a3","resolution":{"observed_at":"2026-08-06T12:21:15.835852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.824638Z","title":"ContextNet: representation and exploration for painting classification and retrieval in context,","venue":null,"work_id":"af808e5f-8dcb-4316-9b90-4783a3346cad","year":2020},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.437235Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:ea13a6d3c64585be74a78ce30e6bf1eb2d1f411b622ba0fd1f7c11d849dce93b","observation_id":"cb2c3577-f29e-4ff7-ba22-6848451ddf9e","resolution":{"observed_at":"2026-08-06T12:21:15.827653Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.815411Z","title":"How to read paintings: semantic art understanding with multi-modal retrieval,","venue":null,"work_id":"2005b5c1-d654-415f-b70f-cc5a292bbda6","year":2018},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.440087Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:0f21e51ea51e902f821d1968b4c638f8215002ba0760ff40ba8323e6b9937171","observation_id":"227ce487-ca6a-4f85-b30b-e37473388c43","resolution":{"observed_at":"2026-08-06T12:21:15.818505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.807352Z","title":"Generating captions for images of ancient artworks,","venue":null,"work_id":"7c16e063-e59c-459e-97e5-b496b121d634","year":2019},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.442949Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:e4466145815fb3132254ea7485973db1bd99688344263698a24cf2f5951ffc7e","observation_id":"b8fde18c-aff2-42f4-b86d-791f7c01a484","resolution":{"observed_at":"2026-08-06T12:21:15.810228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.799009Z","title":"A dataset and baselines for visual question answering on art,","venue":null,"work_id":"0e6c2f2b-e74c-491a-8919-ced488808dce","year":2020},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.445701Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:e78260789e7395459498c33e94edbd74aca9bac0619c5be54ac5f0620b9632b2","observation_id":"803fa69d-df7a-40d3-95bb-2278452f278f","resolution":{"observed_at":"2026-08-06T12:21:15.801853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.790811Z","title":"Iconographic image captioning for artworks,","venue":null,"work_id":"5da9b98c-05a2-427b-a61b-8955a58c606c","year":2021},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.448635Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:1f588d26d15b143a87a3308e321b7ecfb0c130c872937ba557156107b07e3aa5","observation_id":"2528bd1d-25c3-4ea5-97de-16a86f7f7764","resolution":{"observed_at":"2026-08-06T12:21:15.793638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.781635Z","title":"Iconclass: an iconographic classification system,","venue":null,"work_id":"1a0ba35d-8c00-4d5b-ab52-a0b25e3a975f","year":1983},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.451450Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:e50f0dbc54cfd80dfe77b5818636c8f7c98365b6da59465de7f537331ff801a0","observation_id":"f6a71688-807f-4219-9c59-382d2bb1c4fe","resolution":{"observed_at":"2026-08-06T12:21:15.785017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.771952Z","title":"Explain me the painting: Multi-topic knowledgeable art description generation,","venue":null,"work_id":"9572b2ec-bc88-4288-99a2-aaefb9983d01","year":2021},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.454256Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:a2a84109076a43f19ed130843850869c7e92351f7f5bf64cf48c6ba3801aad9c","observation_id":"634affea-ec92-4ce7-874b-3a3714ce6a93","resolution":{"observed_at":"2026-08-06T12:21:15.775091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.00051","last_updated":"2017-04-28T03:53:14Z","snapshot_observed_at":"2026-07-06T05:36:07.769337Z","submitted_at":"2017-03-31T20:39:10Z","title":"Reading Wikipedia to Answer Open-Domain Questions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.00051","snapshot_observed_at":"2026-08-06T12:21:15.457109Z","title":"Reading wikipedia to answer open-domain questions,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.457109Z"},"links":{"cited_paper":"/paper/1704.00051","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:3d738974cb268dbaae0edb8f6918375f8ad26bca7af2e917c811f8db672701b3","observation_id":"b6e747ae-a0df-4e17-a0f2-a48a7065f2b0","resolution":{"observed_at":"2026-08-06T12:21:15.457109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.762817Z","title":"Is GPT-3 all you need for visual question answering in cultural heritage?","venue":null,"work_id":"b9df6b51-c29a-4721-8764-6a17d95b95a4","year":2022},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.460567Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:b13c11f4e014e05b82308c887ee2eb551313bea16d58bc3f94d795784a53e54f","observation_id":"0582f23c-164b-4137-8317-8aa4fe6e72a9","resolution":{"observed_at":"2026-08-06T12:21:15.765957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.753610Z","title":"Exploring the Synergy Between Vision-Language Pretrain- ing and ChatGPT for Artwork Captioning: A Preliminary Study,","venue":null,"work_id":"ff921752-e507-4474-aa78-fc52b400d185","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.463245Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:e8700af9bd5b9640cb4c9de0e84c8693b7a5da3ab8427b46f635058dc163fd7e","observation_id":"cb6d3a00-b468-43c0-9457-842b4bf6981c","resolution":{"observed_at":"2026-08-06T12:21:15.756816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.744627Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale,","venue":null,"work_id":"d03089f5-5dea-4e56-874c-d7785f6e9941","year":2021},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.466019Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:950fca3b98ab3c41344344bd8d6718b991dd756d00f86e5492cc3565c16228d8","observation_id":"854e98c0-f9c8-42c8-af33-75ac959ec5d1","resolution":{"observed_at":"2026-08-06T12:21:15.747936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.736161Z","title":"Flamingo: a visual language model for few-shot learning,","venue":null,"work_id":"3f33d0e2-8d66-455c-b25b-735967aef280","year":2022},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.468917Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:42d393fa0841c72ebc1a243c02a965c18d315a6bd963492b6b77e80d60051fb9","observation_id":"150cff47-c133-499e-b3d4-4e4f410708fe","resolution":{"observed_at":"2026-08-06T12:21:15.739098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.471718Z","title":"Visual instruction tuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.471718Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:678965ec18b0d8bba146cd46122e37b57630bc2c9a3182f24f818995601b985e","observation_id":"c9a50e0a-707c-40ec-aa9b-29fad0d34079","resolution":{"observed_at":"2026-08-06T12:21:15.471718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.722328Z","title":"InstructBLIP: Towards General- purpose Vision-Language Models with Instruction Tuning,","venue":null,"work_id":"4c0347ec-f1ad-4a81-a29a-a74d6200a7d6","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.474586Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:ce93497bcb4e2595058e5e0deba547798371c78bc8c69b7af563f35a5d028f8c","observation_id":"6d0664b4-9995-4025-96fc-bc41d9ab6673","resolution":{"observed_at":"2026-08-06T12:21:15.725332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.713938Z","title":"Reveal: Retrieval- augmented visual-language pre-training with multi-source multimodal knowledge memory,","venue":null,"work_id":"81bf7e04-9f5d-4230-a32c-b217f4481ab6","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.477353Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:2d698241f4ef8087c07a84686ba9fcb58d8e45e0a1b7e8aee0ee04fcafff643f","observation_id":"796d7235-184e-4229-9233-793acc8f5ea4","resolution":{"observed_at":"2026-08-06T12:21:15.717080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.705390Z","title":"EchoSight: Advancing Visual-Language Models with Wiki Knowledge,","venue":null,"work_id":"b60a59cf-e679-4a0b-b15a-82ee24b120ea","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.480040Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:2518611a726699cb92aaa8fbbf8b73f4830580ff298a072ecdda3fd735380b96","observation_id":"e0c4461a-5074-4713-aac2-d48110b8557c","resolution":{"observed_at":"2026-08-06T12:21:15.708566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.696403Z","title":"Wiki-llava: Hierarchical retrieval-augmented generation for multimodal llms,","venue":null,"work_id":"a6be181b-2642-4502-8722-03e4ac7854f2","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.482995Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:ddf44506f5d171c0c270c4330a677ad0ea961e1ce123164e333830643b239ce3","observation_id":"d4765d3a-bea7-46bb-823d-108c18c0938c","resolution":{"observed_at":"2026-08-06T12:21:15.699418Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-08-14T04:17:22.593941Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-06T12:21:15.486031Z","title":"Qwen2. 5-vl technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.486031Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:d05c2924c7d67751517d3122aa82e3ce53c5eaa37becdee4a016849db67dc108","observation_id":"60a1f0ce-cbdd-4d57-87b8-6a4c87f7f94b","resolution":{"observed_at":"2026-08-06T12:21:15.486031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.687514Z","title":"Toolformer: Language Models Can Teach Themselves to Use Tools,","venue":null,"work_id":"8c3d074d-35ab-488c-8261-a5666fb81d38","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.489064Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:8a735047b902766d8de513c2d807712f7ec6dda8d964d0381133ab7b716db005","observation_id":"b1ff6eda-cc9f-404b-bbe8-e9a45320b9d4","resolution":{"observed_at":"2026-08-06T12:21:15.690765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.491934Z","title":"React: Synergizing reasoning and acting in language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.491934Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:de6846bdb4b1df63fd29e8ddff5e9e1fb156fe3479e29b0bf29f849e534dbc87","observation_id":"fcc00fb5-76f2-47b9-bf60-adc2a0cb7931","resolution":{"observed_at":"2026-08-06T12:21:15.491934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.673118Z","title":"Colpali: Efficient document retrieval with vision language models,","venue":null,"work_id":"fe6c1336-bdda-4670-b3e3-352bbc6ae00e","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.494631Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:5f62877cf1d96b6c03babec3eb67305e9e2613bbe2cfc771303615ffba533603","observation_id":"6b1053e5-fd02-4592-9b53-b6675d9f4ca3","resolution":{"observed_at":"2026-08-06T12:21:15.676205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.664209Z","title":"Wikiextractor,","venue":null,"work_id":"65cbee2a-2e77-4ff1-9ec2-2ebce1cc5a2b","year":2012},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.497512Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:68f0ee53c522772ef4a60167666a921eb15717b02656201e66c73ba76bb5e10d","observation_id":"ad6f203a-8928-413e-a0fb-bbc6b3324ca3","resolution":{"observed_at":"2026-08-06T12:21:15.667440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14683","last_updated":"2024-09-23T03:12:43Z","snapshot_observed_at":"2026-08-14T16:43:56.595613Z","submitted_at":"2024-09-23T03:12:43Z","title":"Reducing the Footprint of Multi-Vector Retrieval with Minimal Performance Impact via Token Pooling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.14683","snapshot_observed_at":"2026-08-06T12:21:15.500544Z","title":"Reducing the Footprint of Multi-Vector Retrieval with Minimal Performance Impact via Token Pooling,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.500544Z"},"links":{"cited_paper":"/paper/2409.14683","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:435ca28664136ac1c07a68d7983ba89dd65742541a79b61b3903c39282e6f85e","observation_id":"a4726a7f-1eef-41f6-918b-9b5744df52f2","resolution":{"observed_at":"2026-08-06T12:21:15.500544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.503536Z","title":"Sigmoid loss for language image pre-training,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.503536Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:157126780ada4dc818da8c6efdf45db30f6e3371d252d17468caa55cff81e970","observation_id":"26aa50d7-e60d-4a0c-b09f-33329d07c8c3","resolution":{"observed_at":"2026-08-06T12:21:15.503536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.506829Z","title":"Multi-task learning using uncertainty to weigh losses for scene geometry and semantics,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.506829Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:adc1425502e8a21b18ae36ba0ea580f6a411c53e369c4903b7832abda3ac9b81","observation_id":"8e9eafa2-5e31-410c-a71a-abea7a1ca900","resolution":{"observed_at":"2026-08-06T12:21:15.506829Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.645965Z","title":"Art History: A Preliminary Handbook (1996)","venue":null,"work_id":"d2bf38b4-362a-49d5-9669-9b5097627b31","year":1996},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.509740Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:d02f1b6c610fee0d6aea6625d99b6ddaeacd5d017542a051b3a77345e48800d6","observation_id":"c7b3448b-451d-4bf3-9902-a94a24d34c47","resolution":{"observed_at":"2026-08-06T12:21:15.648818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-13T15:58:13.809876Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T12:21:15.512431Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.512431Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:defbfab8516def4bc296205ae126dc6618e2db476de0d02de920ec89aea414c4","observation_id":"659a6a7a-cd51-4859-9372-5b7321fd9989","resolution":{"observed_at":"2026-08-06T12:21:15.512431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.637108Z","title":"Composed image retrieval using contrastive learning and task-oriented CLIP-based features,","venue":null,"work_id":"63768869-d75f-4509-9cb0-fb1ee88c66f2","year":2023},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.515457Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:685db1ebcaf3fc63f8122d0ba7380dc3b132d1e2ccef87ee38a82e3debe98015","observation_id":"e1c506db-022a-43b2-bcb1-58e6f8c08359","resolution":{"observed_at":"2026-08-06T12:21:15.640222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.628643Z","title":"Artwork interpretation,","venue":null,"work_id":"4b6c5a2a-f2fe-4fd6-8b43-9e1848f009df","year":2022},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.518248Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:c346a2e6262ca649d5f48839f7035cf0514589b3379522ca31d1fbadb7d5c5b7","observation_id":"baa7618f-90fc-4fe1-b022-3a3206541662","resolution":{"observed_at":"2026-08-06T12:21:15.631502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:15.619881Z","title":"Artquest: Countering hidden language biases in artvqa,","venue":null,"work_id":"8f3669ce-b962-4653-a2b3-ada330d5b725","year":2024},"citing_paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:15.520982Z"},"links":{"citing_paper":"/paper/2507.21917"},"observation_digest":"sha256:6fb77c5a10ac48798ccccc5181bf9379f1920c6120355dbb678e5b63f583c1ea","observation_id":"c841696b-c349-4518-a218-08b8bee8ffa7","resolution":{"observed_at":"2026-08-06T12:21:15.623100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.21917","last_updated":"2025-07-29T15:31:58Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T16:18:48.745765Z","submitted_at":"2025-07-29T15:31:58Z","title":"ArtSeek: Deep artwork understanding via multimodal in-context reasoning and late interaction retrieval"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":13,"verified_exact":1,"verified_fuzzy":35},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 4 inbound Pith citation observations for arXiv:2507.21917."}