{"as_of":"2026-08-18T20:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b4e13311f73422544afc5728b32a98c330f1b76fe5b251b0bf65a2c54db68536","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T15:05:57.863886Z","state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T00:02:12.727566Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-10T00:29:47.839616Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"cited_work":{"arxiv_id":"2411.14704","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14704","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"93bf4b2b-de35-4668-a83d-bda0174158d7","year":2024},"citing_paper":{"arxiv_id":"2604.20429","last_updated":"2026-04-22T10:50:38Z","snapshot_observed_at":"2026-08-17T20:26:52.507304Z","submitted_at":"2026-04-22T10:50:38Z","title":"Fast-then-Fine: A Two-Stage Framework with Multi-Granular Representation for Cross-Modal Retrieval in Remote Sensing","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T00:02:12.727566Z"},"links":{"cited_paper":"/paper/2411.14704","citing_paper":"/paper/2604.20429"},"observation_digest":"sha256:35628cc01dfdc2955ad929265b8cf6f1c7dc6321d38d2413d9303e8e5bd24e13","observation_id":"be331e05-aba8-4882-8bdd-c298a7fd228c","resolution":{"observed_at":"2026-05-10T00:29:47.841047Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.14704/citation-record","integrity":"/paper/2411.14704/integrity","json":"/paper/2411.14704/citation-record.json","paper":"/paper/2411.14704"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.607767Z","title":"Remote sensing big data computing: Challenges and opportunities,","venue":null,"work_id":"6e7436d9-351d-4695-8637-e4daf60c0741","year":2015},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.642747Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:e9b912457d3d7ee7734198be3a8902f68bc2bcd1d94416fc22fa24027437b925","observation_id":"4c3c0f33-d76b-44b2-828f-248f4bb5975a","resolution":{"observed_at":"2026-08-12T15:05:58.612459Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.594508Z","title":"Big data for remote sensing: Challenges and opportunities,","venue":null,"work_id":"fc6c6dbe-dbbc-49f6-bc26-c552d83d8f79","year":2016},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.648140Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:eb779fc69fc97477b767b001be263e95d284566d3dd480873a2c0b6f4efdf3bb","observation_id":"b520db60-7253-48a9-ab56-357ceba5605a","resolution":{"observed_at":"2026-08-12T15:05:58.599090Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.578302Z","title":"Understanding urban landuse from the above and ground perspectives: A deep learning, multi- modal solution,","venue":null,"work_id":"6c9c1fa7-fc18-42e8-8c70-6944f6d0d316","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.653137Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:8ccb9d86de553e9760b5896a6b494c79f39f8b59690d601e5558ccb15efa77d2","observation_id":"186ad7d1-a193-4143-a9b6-b76fada333c3","resolution":{"observed_at":"2026-08-12T15:05:58.583419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.564573Z","title":"Hyperspectral data analysis for arid vegetation species: Smart & sustainable growth,","venue":null,"work_id":"1075291d-9195-4664-8c0b-2323edae402a","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.660856Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:b395223237dc336737fe2c144717b9baaf946fa23c8629cd918b8978a45785be","observation_id":"43844814-7191-468d-a558-7acd3ab66674","resolution":{"observed_at":"2026-08-12T15:05:58.569310Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.549429Z","title":"Google earth engine cloud computing platform for remote sensing big data applications: A comprehensive review,","venue":null,"work_id":"d9b39105-ab50-481c-9db5-f8fd36e7231e","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.665468Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:40d3422046a3bde4ea0108244490b47d4889ec160baedf7e0a4fe0f695211d1e","observation_id":"6fa11a44-fe85-41c9-a49a-5b0c45c4ce5f","resolution":{"observed_at":"2026-08-12T15:05:58.554913Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.669876Z","title":"Nwpu- captions dataset and mlca-net for remote sensing image captioning,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.669876Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:067d394abb4c3ef1d2e12077ee4eac563b9fcda15544f97e306127b39cc34c42","observation_id":"960bfcb1-ca03-4497-8df8-27754115cfda","resolution":{"observed_at":"2026-08-12T15:05:57.669876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.516916Z","title":"Textrs: Deep bidirectional triplet network for matching text to remote sensing images,","venue":null,"work_id":"bcb69c45-7878-4c9a-a2f6-e9ca8e320772","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.674957Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:906c21432dad377fb079849060e77b20f2296691ae00a21036ca951e8a803e93","observation_id":"654ef5dd-7f5b-4121-bd63-54b3ae71712e","resolution":{"observed_at":"2026-08-12T15:05:58.521644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.679755Z","title":"A deep semantic alignment network for the cross-modal image-text retrieval in remote sensing,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.679755Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:a20cd672cf5af24c400e3a1d454ca176ea03a30eb25ff161823c546d04ddce65","observation_id":"313642ac-f437-493f-823a-de51ef14e235","resolution":{"observed_at":"2026-08-12T15:05:57.679755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.494798Z","title":"Fusion-based correlation learning model for cross-modal remote sensing image retrieval,","venue":null,"work_id":"eb96805f-ede1-43df-b506-c46b2b5aebf3","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.683688Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:d09fa3ffa3d8a531addc5517f2ef015acda052e61a0bc26d66ba3fab4a72fdd5","observation_id":"0a042db3-5854-4d0e-9c98-8f731f04820f","resolution":{"observed_at":"2026-08-12T15:05:58.499246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.481855Z","title":"Cross spectral image reconstruction using a deep guided neural network,","venue":null,"work_id":"e5518036-7646-4035-a66e-e83528ee2563","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.687271Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:2c6ada8406939f26a461cfaa6de83bb3b4a5754e9bb28b7370f65aa426f86779","observation_id":"b1447892-a6cf-4ffb-84d5-2d9dfe23f503","resolution":{"observed_at":"2026-08-12T15:05:58.486375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.465277Z","title":"Image super-resolution using t-tetromino pixels,","venue":null,"work_id":"d3627432-4bb6-4e45-bf7b-ed46b4fdddbe","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.691404Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:3749175d8e02a0af08d5c148b87f939b75631a984036dde6768d1b1a742f1193","observation_id":"f87b27fd-5a3f-4f20-b745-98abb2d47e4f","resolution":{"observed_at":"2026-08-12T15:05:58.472800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.695460Z","title":"Exploring a fine-grained multiscale method for cross-modal remote sensing image retrieval,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.695460Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:39ad7ae7772ec65be220eb50b6411bf7738e828f24d638e975125fe7898045fb","observation_id":"a2bafe0a-f97a-414e-9580-6e01c0556ead","resolution":{"observed_at":"2026-08-12T15:05:57.695460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.699844Z","title":"Remote sensing cross-modal text-image retrieval based on global and local information,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.699844Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:f2b279fdc4c2c6ed71e68d04bc13a769d2d55d1355e7d9bde3ccca3c6d8f3559","observation_id":"deb1efde-12a0-4211-994d-3d1cc98e2126","resolution":{"observed_at":"2026-08-12T15:05:57.699844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.436486Z","title":"A lightweight multi-scale crossmodal text-image retrieval method in remote sensing,","venue":null,"work_id":"120d1daf-b135-4bd4-ac3b-36f749c9205b","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.704005Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:710744610f92cf79a0cf1f75bfbb4c3cb3c9e39a779c4c7c31c36ed9f674e988","observation_id":"14cdfa4f-a120-4b0e-ae72-c3c707dcabf0","resolution":{"observed_at":"2026-08-12T15:05:58.440911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.422609Z","title":"Exploring uni-modal feature learning on entities and relations for remote sensing cross-modal text-image re- trieval,","venue":null,"work_id":"c0cab0b8-e3f6-4852-b6e9-d360edf1214f","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.707863Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:daea7230b460504e4c8b5f022aa90a10cf32f81e687c0d544120f0b1f1756d34","observation_id":"c8db1e29-8418-4854-ae4c-169435be8548","resolution":{"observed_at":"2026-08-12T15:05:58.427872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.409098Z","title":"Hypersphere-based remote sensing cross-modal text-image retrieval via curriculum learning,","venue":null,"work_id":"bedb3578-a6ae-4b4f-b16b-4729279c61ed","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.711725Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:fc155608570b76afd07401609399db9f3e16dbdf40ea61a0c7dd950830b49126","observation_id":"70b6b766-acec-470d-b78a-489c3a11e6af","resolution":{"observed_at":"2026-08-12T15:05:58.413683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-16T09:25:53.087782Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-12T15:05:57.715976Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.715976Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:89fc37addd472f8757cda63f5858bbec50182848c1185fb890fd8a705872e0f4","observation_id":"d19e02e5-e545-43e2-b5ad-b9799914777c","resolution":{"observed_at":"2026-08-12T15:05:57.715976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.394847Z","title":"Long short-term memory recurrent neural network architectures for large scale acoustic modeling,","venue":null,"work_id":"f713e819-382c-4aff-aa25-cc38e421c731","year":2014},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.720276Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:17e5896960e9b35eeeb6e4e334996e29c7778c6f7e85f36fb5555e44c41d49ee","observation_id":"4f20a153-5c0f-4009-a05c-947e74b345ca","resolution":{"observed_at":"2026-08-12T15:05:58.399313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1412.3555","last_updated":"2014-12-11T06:46:53Z","snapshot_observed_at":"2026-08-13T10:35:27.214652Z","submitted_at":"2014-12-11T06:46:53Z","title":"Empirical Evaluation of Gated Recurrent Neural Networks on Sequence Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.3555","snapshot_observed_at":"2026-08-12T15:05:57.724413Z","title":"Empirical evalua- tion of gated recurrent neural networks on sequence modeling,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.724413Z"},"links":{"cited_paper":"/paper/1412.3555","citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:e33e1907e161b1347508124cd3f0662eac61047e2b484cc9dd9374c772fe04a6","observation_id":"beb708e3-24d7-4cce-ae2e-c8810b2f1c7d","resolution":{"observed_at":"2026-08-12T15:05:57.724413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.728685Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.728685Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:4fb62a4f02498e4af601f546053485fd4154ff3cac472d42d97c1785816e5f8f","observation_id":"6dce598e-f33e-42cb-8ffc-3cf4bb0e6501","resolution":{"observed_at":"2026-08-12T15:05:57.728685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.373568Z","title":"Multiscale salient alignment learning for remote sensing image-text retrieval,","venue":null,"work_id":"e97334ea-4945-4e26-a4f7-9e9767048db9","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.732957Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:80da6f99d3b18d5175d0771dcf5ccf0cf79b31e59b49d40187d0a2ac1824457f","observation_id":"dce4e68e-4451-4a12-980a-ede130182e20","resolution":{"observed_at":"2026-08-12T15:05:58.378152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.359007Z","title":"Interacting- enhancing feature transformer for cross-modal remote sensing image and text retrieval,","venue":null,"work_id":"fb43b4e8-47ca-467a-84a9-5ca9455dd452","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.736707Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:102fab525ae25260cfa990a8bd6b2ee76b784fb4817c8ed44f76c2be2fa5e1c8","observation_id":"71c63976-4336-441b-8ff7-2f40c3b50181","resolution":{"observed_at":"2026-08-12T15:05:58.364329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.342688Z","title":"Align before fuse: Vision and language representation learning with momentum distillation,","venue":null,"work_id":"daf655d9-d416-4c67-b440-f53f89ca8484","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.740423Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:26b0717afe1ab97d4ffdc81abce46ee8f74706f338abe93d8ffabd36abf2c4aa","observation_id":"9465be24-0dca-4812-a1c8-70fdddfadfc5","resolution":{"observed_at":"2026-08-12T15:05:58.349930Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.327586Z","title":"Deep saliency smoothing hashing for drone image retrieval,","venue":null,"work_id":"6a379627-6175-4b1a-a4f1-1e9fc6020333","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.744193Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:760460109d2ba1d44d31569b5bb8484a5f1aced33605867ce4896d69134ee2c7","observation_id":"e6a07d92-cebb-495a-8826-6c0c53a5ed8c","resolution":{"observed_at":"2026-08-12T15:05:58.332354Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.310177Z","title":"Multitask learning for sar ship detection with gaussian-mask joint segmentation,","venue":null,"work_id":"a3153b2c-bbbc-4cc9-8e8d-f94acf9ccaa7","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.748050Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:9c868cd085383673f922045ed59a4551cb2fab04587ee18acc1977e8318f6a7d","observation_id":"76ed2888-dad4-4d5a-9698-91351679a81c","resolution":{"observed_at":"2026-08-12T15:05:58.314377Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.297962Z","title":"Swin transformer: Hierarchical vision transformer using shifted windows,","venue":null,"work_id":"fee7a107-37db-4861-9bab-38cfe6f337d8","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.751703Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:7a261703fb69bb5c9c8f2702b48065e32cc94234f1c70f2f03c3a7daf8949d26","observation_id":"930a1883-858a-468c-ae0d-e8757f9c2a18","resolution":{"observed_at":"2026-08-12T15:05:58.302154Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.285398Z","title":"Global context vision transformers,","venue":null,"work_id":"98cb2788-07d2-4483-a87b-c6ce409971d1","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.755719Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:e11385f03d04ce17e95b4798c31b2ad123b5f8ba8d7c635e015ce53215019c09","observation_id":"2380a155-4554-40be-b0f7-8cda192742e4","resolution":{"observed_at":"2026-08-12T15:05:58.289581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.272443Z","title":"Matching images and text with multi-modal tensor fusion and re- ranking,","venue":null,"work_id":"9c5cbf1b-22d7-4d4a-a832-a30426490d22","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.759505Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:650813248fc761c434f0388ffca43647cc00efe7b56ca52ac2c57e7add487ed1","observation_id":"37aa4e64-5910-4588-9fbb-cf1ea37451dc","resolution":{"observed_at":"2026-08-12T15:05:58.276858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.259617Z","title":"Vse++: Improving visual-semantic embeddings with hard negatives,","venue":null,"work_id":"d2389990-e818-4c30-a041-f66cf1104a26","year":2017},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.763722Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:a0e82076f3a9d7403d1d0c476b55d7dad417acba30531900e98d9de09b40e125","observation_id":"6d79f654-7652-4f77-a091-b4447c4ddbfb","resolution":{"observed_at":"2026-08-12T15:05:58.264258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.767188Z","title":"Exploring models and data for remote sensing image caption generation,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.767188Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:8a98d03482b86afc8484141046a2a4545d58320037da6cfd231550253f6e69a8","observation_id":"84ac96ff-b178-43ae-82b3-893433494fb9","resolution":{"observed_at":"2026-08-12T15:05:57.767188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.236889Z","title":"Deep semantic understanding of high resolution remote sensing image,","venue":null,"work_id":"ffd4bd9b-98a9-4b47-876b-d12733fa7d46","year":2016},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.770808Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:f02a1c299f378b51abf2e29d50af14ad2dfeb2adb2683d5d77390037ec4f069f","observation_id":"fafb1176-24c3-427d-b2a1-e7c343df78e3","resolution":{"observed_at":"2026-08-12T15:05:58.242536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.220557Z","title":"End-to-end convolutional semantic embeddings,","venue":null,"work_id":"f7927f39-8dca-402b-8ced-9a11fdcddd2b","year":2018},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.774534Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:9c15a0f0b74223e9b98babfb4638445c4d9975377a3293b4d998e3206f349466","observation_id":"44d0bb0b-68d2-4af5-bf61-64699935b562","resolution":{"observed_at":"2026-08-12T15:05:58.225883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.205158Z","title":"Cross-modal semantic correlation learning by bi-cnn network,","venue":null,"work_id":"a56cfe1f-47c2-4e44-99a5-ea42d1dccf4a","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.778748Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:1487eacb083a72293d209254b040ab7b653f561cf0ca90f495c2a6929c0d7774","observation_id":"519cb981-5d09-44c6-896e-a8e9aecc3cd6","resolution":{"observed_at":"2026-08-12T15:05:58.210444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.190767Z","title":"Dual-path convolutional image-text embeddings with instance loss,","venue":null,"work_id":"8a611450-3c4f-4444-8c27-9e47c0619f31","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.782641Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:da1ed1a3866a389ef4e6d872248d2de578c6893a2e737988f3e3bd31676013d3","observation_id":"f110aff3-2c77-467e-856d-caf394a2b75a","resolution":{"observed_at":"2026-08-12T15:05:58.195285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.786515Z","title":"Deep supervised cross-modal retrieval,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.786515Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:8b307e3f1220fcf11548992e02595b555365e9d9aff0952d13e6c16d3eb6bd7e","observation_id":"e1216bf4-dd3e-4a46-a799-90b70d2aa7f3","resolution":{"observed_at":"2026-08-12T15:05:57.786515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.165509Z","title":"Learning semantic concepts and order for image and sentence matching,","venue":null,"work_id":"9ac638d8-0d58-4ed1-a1e8-66991a749d08","year":2018},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.790384Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:6c8d6935ced38fcfc830ee54ce49fe16e16f5f7c599864a4bf70a4dc3fa1a25f","observation_id":"03dc5032-001b-4cd2-9d2d-27ef02950545","resolution":{"observed_at":"2026-08-12T15:05:58.170650Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.151060Z","title":"Stacked cross attention for image-text matching,","venue":null,"work_id":"d021d660-aaef-4269-af43-263a2c1d4796","year":2018},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.794558Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:221962300bf98eb8b85d8a02acc190aed9fec14130bcf64911cb436c7a2297c3","observation_id":"ee5e49a4-9c8a-4ed9-be2d-5070c4577c11","resolution":{"observed_at":"2026-08-12T15:05:58.156138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.138030Z","title":"Cross- modal attention with semantic consistence for image–text matching,","venue":null,"work_id":"2fadf9fa-31a4-48a5-8af4-b6db11f750a6","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.798448Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:7f7bd5d6777310215dd0344a8b5bac871012739ce491177f7f20d18f9734c990","observation_id":"8fd228ad-e85b-43c2-a565-2c2e4aa3d91a","resolution":{"observed_at":"2026-08-12T15:05:58.142074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.125323Z","title":"Visual semantic reasoning for image-text matching,","venue":null,"work_id":"1ea83aa3-014f-478d-be92-dbf13943aee4","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.802192Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:290bfef973f2ab8fd9dd15e42af9a67caab9a8b1f2dcb91df0800f45da2ec094","observation_id":"38ae2995-86c3-4b14-93b2-da73ff3a6f7f","resolution":{"observed_at":"2026-08-12T15:05:58.129587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.110281Z","title":"Image-text embedding learning via visual and textual semantic reasoning,","venue":null,"work_id":"1416bc64-2da6-4d83-b3b4-979685549c14","year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.805867Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:ae459aba507246f48bc323497fcbe29335fc63bd6bab925e51fdd51912b67341","observation_id":"91c20eb1-1199-455f-a128-3099ece39e98","resolution":{"observed_at":"2026-08-12T15:05:58.115153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.809534Z","title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.809534Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:c4dcb7f7b9dd3663b33c04db902752d5f64a92de4bc269b999f96ca7503b97c3","observation_id":"1414cfaf-f4aa-449e-a1a5-f582b8bb01c7","resolution":{"observed_at":"2026-08-12T15:05:57.809534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.088823Z","title":"Lxmert: Learning cross-modality encoder representations from transformers,","venue":null,"work_id":"313b043c-e4bd-4613-b889-72722d8caae0","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.813277Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:e0105b175338cdf7076963c8b752e82f84c65acaf8d88ed2b850b5919b377196","observation_id":"ecf46abb-c98e-4236-af91-6bb7d5159778","resolution":{"observed_at":"2026-08-12T15:05:58.093662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.075571Z","title":"Fashionbert: Text and image matching with adaptive loss for cross- modal retrieval,","venue":null,"work_id":"fc46efbb-b0a0-477b-86bd-f7ab1b4b4138","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.816997Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:d82b217603ba3d27d34b80ce47eb5876d665c84fbef0325e77f6cf89ac183e04","observation_id":"c1a80255-d52c-4e74-87fb-1c654dfcc07f","resolution":{"observed_at":"2026-08-12T15:05:58.080079Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.062521Z","title":"Learning the best pooling strategy for visual semantic embedding,","venue":null,"work_id":"320c26eb-158d-4502-b241-02bbed3e9058","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.820850Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:c5bee39ccf6454ea449c1e47127a3f362d25ba4d1975ec5547bc43722d2f2d1c","observation_id":"84f014d7-3af0-4263-b883-94ef99a55daa","resolution":{"observed_at":"2026-08-12T15:05:58.066838Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.00849","last_updated":"2020-06-22T09:09:22Z","snapshot_observed_at":"2026-08-15T14:49:08.287205Z","submitted_at":"2020-04-02T07:39:28Z","title":"Pixel-BERT: Aligning Image Pixels with Text by Deep Multi-Modal Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.00849","snapshot_observed_at":"2026-08-12T15:05:57.824935Z","title":"Pixel-bert: Aligning image pixels with text by deep multi-modal transformers,","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.824935Z"},"links":{"cited_paper":"/paper/2004.00849","citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:f77c5db972c8796c9356d4cc2f1a801dfba8aa254bee12fc16a0a5ff3307db4b","observation_id":"0e90948c-19b9-4e50-8e10-f5988d1ed44f","resolution":{"observed_at":"2026-08-12T15:05:57.824935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.829050Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.829050Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:940eb8c7d569d9debefc66a1c2fba46ceff5f4a1b4aa45c2f059d843432d3c48","observation_id":"bf64ed5d-685c-43e1-9734-415474ed2ece","resolution":{"observed_at":"2026-08-12T15:05:57.829050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.037721Z","title":"Vista: Vision and scene text aggregation for cross-modal retrieval,","venue":null,"work_id":"9ac05722-6a40-4822-a2b1-c2a473f353f1","year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.832696Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:4ff3c5031fddc291a9146574b8799e838c172296121cefeba9b49656a9530a5d","observation_id":"e2c38548-073c-429f-a7d2-c76397f0bc7a","resolution":{"observed_at":"2026-08-12T15:05:58.043211Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.022875Z","title":"Vilt: Vision-and-language transformer without convolution or region supervision,","venue":null,"work_id":"189c5315-e747-4766-8d58-946278e7d254","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.836567Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:cc03d348125648b763db5aab03a0635e94d84a24e326f0a5f91ed8eba2f8d4c5","observation_id":"d6f8ddff-9a5c-43b6-82b6-5bc90cca8ffb","resolution":{"observed_at":"2026-08-12T15:05:58.027761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.008687Z","title":"Vlmo: Unified vision-language pre-training with mixture-of-modality-experts,","venue":null,"work_id":"ae3f342f-a34a-4fb1-acae-407294eb1e17","year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.840435Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:fb6c3b71253da3bdb0af4f24af8623436a732894a9f8f9c2c7e3f5d0a0592528","observation_id":"597fa9db-e019-4720-8555-7da321fee29a","resolution":{"observed_at":"2026-08-12T15:05:58.013236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.991591Z","title":"Knowledge-aided momentum contrastive learning for remote-sensing image text retrieval,","venue":null,"work_id":"180fefb7-2b82-466e-9907-ef97de88fbb2","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.844283Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:185b4629e3573a1952cb6edc9cf9131d562b1aebe8d9d441447a87210a0b29ca","observation_id":"45903d78-eded-4900-a047-c4855e2cfac6","resolution":{"observed_at":"2026-08-12T15:05:57.999616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.978437Z","title":"Multi- scale interactive transformer for remote sensing cross-modal image-text retrieval,","venue":null,"work_id":"f97777ea-e345-456b-bb14-ac20d28d7679","year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.848175Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:a585aae7cd4d4095b0dcb279fe4158a2d468af4e7bb1eeb047409575aa0997a5","observation_id":"9b3afb81-747a-406f-858f-61a9b7dd3591","resolution":{"observed_at":"2026-08-12T15:05:57.983362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.852135Z","title":"Parameter-efficient transfer learning for remote sensing image-text retrieval,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.852135Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:6cdab64ea7c09e7e1543dedef9909efc329569fec1ecb92ad8ffe25e34489cf2","observation_id":"87a56034-0031-4957-9722-3c6a38a28750","resolution":{"observed_at":"2026-08-12T15:05:57.852135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.856091Z","title":"Integrating multisubspace joint learning with multilevel guidance for cross-modal retrieval of remote sensing images,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.856091Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:4e248718da2ee13a02fdf0a6839541dde7573d5280b554df00a047d5dec3e7be","observation_id":"8e1b2a61-fe9a-4da8-bcee-5a7041d2c5ce","resolution":{"observed_at":"2026-08-12T15:05:57.856091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.946840Z","title":"Bert: Pre-training of deep bidirectional transformers for language understanding,","venue":null,"work_id":"99380dc9-ab78-41b2-b335-cd5f9eb81639","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.859812Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:1063219bc3285433f5cdb04f61d82c3086c6cc88721aeb28db9950919908c025","observation_id":"1ac19980-e07b-4711-8c32-64bc5507a56c","resolution":{"observed_at":"2026-08-12T15:05:57.951457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.931980Z","title":"Momentum contrast for unsupervised visual representation learning,","venue":null,"work_id":"f1a4242f-b0b7-438c-a728-7e37727b7cd3","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.863886Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:0d180aff5e3207cb13df37ccb7a2fbcc20c4ff2939372c9efd4e7990e9184381","observation_id":"852a2f44-d06f-416e-843f-3fb4879354b6","resolution":{"observed_at":"2026-08-12T15:05:57.937860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-15T20:39:19.485563Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":0,"verified_fuzzy":41},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 1 inbound Pith citation observation for arXiv:2411.14704."}