{"as_of":"2026-08-10T06:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1b322de00656bb4e1e4b1499d2c390500341b4286bb44513c90760baac4764fa","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:00:54.910555Z","state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.03709/citation-record","integrity":"/paper/2506.03709/integrity","json":"/paper/2506.03709/citation-record.json","paper":"/paper/2506.03709"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.539630Z","title":"Syn2real domain generalization for underwater mine-like object de- tection using side-scan sonar","venue":null,"work_id":"5d588c0f-d0ba-4cb7-be5d-b09668f27788","year":2025},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.740031Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:9a08f90a4074477aaaef3daed14b2153a239c2d36ab4fed6686f8f75ea217618","observation_id":"61bcb3ed-c11e-41de-9292-9953c27c056d","resolution":{"observed_at":"2026-08-07T11:00:55.544548Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.518704Z","title":"What a mess: Multi-domain evaluation of zero-shot semantic segmentation","venue":null,"work_id":"4278b932-db3b-42e4-9d20-c8d6b0e488c6","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.745956Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:bb23a8d7cb8a26999f7d1ef952d26972616f0cc51ca26f1235e78ab98bf0760e","observation_id":"76eaad89-6c37-4ea9-b870-15590b3a0629","resolution":{"observed_at":"2026-08-07T11:00:55.527529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.500418Z","title":"Coco- stuff: Thing and stuff classes in context","venue":null,"work_id":"815e8e44-1410-46e3-ab02-aef249596f26","year":2018},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.750519Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:071da001d75df2a26a6a090e7bf5a0c69e9af71dba612f8fc94679822a3b523d","observation_id":"f1354a32-ad22-4205-8099-8c7d5cd78189","resolution":{"observed_at":"2026-08-07T11:00:55.505663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.755631Z","title":"Cat- seg: Cost aggregation for open-vocabulary semantic seg- mentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.755631Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:4bdf17963db5f85d37327081781eb081c711228ae76b125b2a08feda574a1f2e","observation_id":"c040cc88-6947-4f53-9256-b894433ba3e1","resolution":{"observed_at":"2026-08-07T11:00:54.755631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.470595Z","title":"De- coupling zero-shot semantic segmentation","venue":null,"work_id":"f5093db3-55b2-473a-abaa-913812dfb95f","year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.760788Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:d579c4e369f29c99695705e8340f983f645bf686f554564357a497c8353a6b54","observation_id":"8225a432-8fb9-4c19-862f-d9d99a7f709f","resolution":{"observed_at":"2026-08-07T11:00:55.475641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.765166Z","title":"Open- vocabulary panoptic segmentation maskclip","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.765166Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:d86f561d250825c8f5920c9af05dfa1a1b4b250d1eb3e37a9a4279da185d67bf","observation_id":"e34b6372-d23f-449d-9ade-7634d1f958e0","resolution":{"observed_at":"2026-08-07T11:00:54.765166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.442077Z","title":"The pascal visual object classes challenge: A retrospective","venue":null,"work_id":"186a5517-e262-461a-ba56-4739f58f6170","year":2015},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.770060Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:110da3940b7898412fb15ccb6309b1d77eba6d5db214232534e50b5f15f2b5d8","observation_id":"d2defca1-eca3-4294-8507-7965987b4cf6","resolution":{"observed_at":"2026-08-07T11:00:55.447014Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.425096Z","title":"Scal- ing open-vocabulary image segmentation with image-level labels","venue":null,"work_id":"de8acfef-18be-4ccf-b2d9-12bdedf4caea","year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.774420Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:197175df93def10454a5d626934fa1d1d5f41c82a550845f52a5c8bccb6caf14","observation_id":"9b8589ed-ec1e-42b0-978d-9ee7e4c0c37f","resolution":{"observed_at":"2026-08-07T11:00:55.430009Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.402405Z","title":"Multispectral video se- mantic segmentation: A benchmark dataset and baseline","venue":null,"work_id":"cbcee5d1-048b-4305-9f28-37f3956824a6","year":2023},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.778848Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:e5643d7c99db609d5ca9b0b4fdf1db555ac95caea5596cfcbb09bc45bc4e9ca4","observation_id":"e6b3209c-6663-4ac3-a950-a2d85c81a5be","resolution":{"observed_at":"2026-08-07T11:00:55.413145Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.783608Z","title":"Scaling up visual and vision-language representa- tion learning with noisy text supervision","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.783608Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:3169ef00d2046bd605b984294b5dfc158888dfd75784f3ae35cacec53e10b8a8","observation_id":"06b53928-1b0a-4d3b-bc2f-cd6a3eee769d","resolution":{"observed_at":"2026-08-07T11:00:54.783608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.373642Z","title":"An efficient approach with dynamic multiswarm of uavs for forest firefighting","venue":null,"work_id":"04d0f5e3-a0f7-408a-9d35-73f7509b2931","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.788414Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:56e9b33d6d43eb0ebc2ce6876c12b23d1b517141a3abe19715735c46c35961fd","observation_id":"d7709b8c-c5ef-4dcd-a154-628bb19fb673","resolution":{"observed_at":"2026-08-07T11:00:55.378331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.356712Z","title":"A resource-efficient decentralized sequential planner for spa- tiotemporal wildfire mitigation","venue":null,"work_id":"1222749b-0168-468e-8495-ff7779f48cb3","year":2025},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.792908Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:ea5f10137b7b8d1b2c5cbd445bc68db4fe5787a3d4c7bb5994786a3b587e5cf3","observation_id":"b71fb33b-2ef7-4282-a8eb-dc0dad063e5d","resolution":{"observed_at":"2026-08-07T11:00:55.362057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.10054","last_updated":"2022-02-21T09:03:34Z","snapshot_observed_at":"2026-08-09T11:23:03.445810Z","submitted_at":"2022-02-21T09:03:34Z","title":"Fine-Tuning can Distort Pretrained Features and Underperform Out-of-Distribution","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.10054","snapshot_observed_at":"2026-08-07T11:00:54.797427Z","title":"Fine-tuning can distort pretrained fea- tures and underperform out-of-distribution","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.797427Z"},"links":{"cited_paper":"/paper/2202.10054","citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:9fde1f3e439e0a9bb14bc0f21f16d979aace5da0244c8fc5806656707cb4fbe4","observation_id":"3b7a43fd-23ec-47e4-bee7-cee29268ed97","resolution":{"observed_at":"2026-08-07T11:00:54.797427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.338122Z","title":"Caltech aerial rgb-thermal dataset in the wild","venue":null,"work_id":"3b696df5-da0f-48d5-a084-214bb73ec595","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.802076Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:9ddd40d5aa3cb54ce13df4af9116a806f66f749fce25610514c22624762e47e4","observation_id":"d925f3fa-41e5-42e1-b517-93bc93e31d41","resolution":{"observed_at":"2026-08-07T11:00:55.342852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.321652Z","title":"Open-vocabulary semantic segmentation with mask-adapted clip","venue":null,"work_id":"f9137dbd-f246-453f-bfd8-35dd397d0a8c","year":2023},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.806584Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:17ebff224b7cbb7bc7789c4f4a55d2b2c5799e0a6935791a0b1a32319703d545","observation_id":"26dfce84-5c02-4655-82dd-55a0c1e62a1e","resolution":{"observed_at":"2026-08-07T11:00:55.326883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09110","last_updated":"2023-10-01T21:44:23Z","snapshot_observed_at":"2026-08-01T19:14:56.803459Z","submitted_at":"2022-11-16T18:51:34Z","title":"Holistic Evaluation of Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.09110","snapshot_observed_at":"2026-08-07T11:00:54.811123Z","title":"Holistic evalu- ation of language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.811123Z"},"links":{"cited_paper":"/paper/2211.09110","citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:071f511388687dbf505e10294f6ee856236d03d7d5b3cf07cd19d3f0374c2908","observation_id":"734ea006-f8da-425d-8a43-001721b94a15","resolution":{"observed_at":"2026-08-07T11:00:54.811123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.306240Z","title":"Uavid: A semantic segmentation dataset for uav imagery","venue":null,"work_id":"e43a1b0c-200d-457c-bc10-dc582e87d44f","year":2020},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.816141Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:6766cb2d231e34e65139b29d9c4a3ec1c8b8dc3b0e6ebcdb35a6e012c54d47ad","observation_id":"6df921a1-10ad-4f78-a4fc-df9e1ce36f49","resolution":{"observed_at":"2026-08-07T11:00:55.310869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.820409Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.820409Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:ab5d42846b81795ea7b020eceb530820be9f884b2ec769ef3e3533f32867cc53","observation_id":"1986572b-a1de-4181-ad50-be376089ffeb","resolution":{"observed_at":"2026-08-07T11:00:54.820409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16067","last_updated":"2024-07-22T21:54:19Z","snapshot_observed_at":"2026-07-06T18:50:22.540341Z","submitted_at":"2024-07-22T21:54:19Z","title":"LCA-on-the-Line: Benchmarking Out-of-Distribution Generalization with Class Taxonomies","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.16067","snapshot_observed_at":"2026-08-07T11:00:54.824912Z","title":"Lca-on-the-line: Benchmarking out-of-distribution generalization with class taxonomies","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.824912Z"},"links":{"cited_paper":"/paper/2407.16067","citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:e9872bf804e388d2a345e95b501204697bf6795e85c58edf64ed483c82be2166","observation_id":"067a8bb2-80b0-42a1-9339-c6959c4d70c3","resolution":{"observed_at":"2026-08-07T11:00:54.824912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.829772Z","title":"Fully complex-valued fully con- volutional multi-feature fusion network (fc 2 mfn) for build- ing segmentation of insar images","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.829772Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:cc747ccb39a205566ed0396f88705295fe6ee97da091543c5a92e01aeaefbd6f","observation_id":"62af2453-9620-4cfe-a25e-31f460f59953","resolution":{"observed_at":"2026-08-07T11:00:54.829772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.261381Z","title":"Deepmao: Deep multi-scale aware over- complete network for building segmentation in satellite im- agery","venue":null,"work_id":"dbc2ade5-7208-4dc1-9d58-50a128de11b7","year":2023},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.835668Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:4b43f21c7ef455cc2a86e6a29cca560bc101615c10073c67d893d73a1bebf083","observation_id":"4ccb78e6-1308-460d-8d51-125ba0a0f238","resolution":{"observed_at":"2026-08-07T11:00:55.267194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.243540Z","title":"Ssl-rgb2ir: Semi-supervised rgb-to-ir image-to-image translation for enhancing visual task train- ing in semantic segmentation and object detection","venue":null,"work_id":"34a89bb0-a484-418e-b5db-0c6297563088","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.840306Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:fa6acceb8bf80ae395b7f45893b73465b1eb1689aa06955e63ec24d33c6f4df2","observation_id":"6e874028-ec58-4e5d-add3-6f0fcd8580c3","resolution":{"observed_at":"2026-08-07T11:00:55.248554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.227038Z","title":"Skd- net: Spectral-based knowledge distillation in low-light ther- mal imagery for robotic perception","venue":null,"work_id":"1bad1597-ebd1-4e6e-b01a-d8f6cd22ef1e","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.844861Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:a3b6c7846a48aef28e35be003bc308ad8efccb295974ec0eda0f20e2c2bbff05","observation_id":"769d20ca-7863-4742-95e1-78745631a294","resolution":{"observed_at":"2026-08-07T11:00:55.232254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15728","last_updated":"2025-04-22T09:22:11Z","snapshot_observed_at":"2026-08-07T16:00:18.165129Z","submitted_at":"2025-04-22T09:22:11Z","title":"SAGA: Semantic-Aware Gray color Augmentation for Visible-to-Thermal Domain Adaptation across Multi-View Drone and Ground-Based Vision Systems","version":1},"cited_work":{"arxiv_id":"2504.15728","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.15728","snapshot_observed_at":"2026-08-07T11:00:54.969171Z","title":"SAGA: Semantic-Aware Gray color Augmentation for Visible-to-Thermal Domain Adaptation across Multi-View Drone and Ground-Based Vision Systems","venue":"cs.CV","work_id":"5f1f6e26-c77d-4f40-8ac2-0c57e033721c","year":2025},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.849654Z"},"links":{"cited_paper":"/paper/2504.15728","citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:fe3d9838faa95a606a94ff63bd43de95b716436dc392ea5d470c199e3f76f4aa","observation_id":"8cc2452f-7b70-46a4-b991-7dde6f189603","resolution":{"observed_at":"2026-08-07T11:00:54.976223Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.209872Z","title":"Ogp- net: Optical guidance meets pixel-level contrastive distilla- tion for robust multi-modal and missing modality segmen- tation","venue":null,"work_id":"fb8ef95e-b6ac-4a1e-8f68-acb172fa8013","year":2025},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.854430Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:c2c5a4e7dcc224ebbcbbe117b356a166059844b2ea70d23b50e5b632f8794fa4","observation_id":"8ce25091-7025-4975-9e0b-153017e6e999","resolution":{"observed_at":"2026-08-07T11:00:55.215098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.191934Z","title":"Isprs potsdam dataset within the isprs test project on urban classification, 3d building reconstruction and semantic labeling, 2012","venue":null,"work_id":"a481a4b7-1c30-4925-9d3b-203b27fa6f08","year":2012},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.858663Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:3ef999fbc1aac6a1c8508d095005e24664b90bd3bff3eb746a743efb93d9c5f6","observation_id":"ee30a712-b673-4314-8d98-5b34950791ca","resolution":{"observed_at":"2026-08-07T11:00:55.197910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.175587Z","title":"Piafusion: A progressive infrared and visible im- age fusion network based on illumination aware.Information Fusion, 83:79–92, 2022","venue":null,"work_id":"aaa7ff63-887f-4af2-a6cc-94a0305ae88a","year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.862947Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:55a037863e6d8d6bcc30acea20dca44dac0f15a43b355758777fa04368170c00","observation_id":"cb27bcad-15a0-4ec2-960f-e1b7af728c17","resolution":{"observed_at":"2026-08-07T11:00:55.180836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.867104Z","title":"Mrfp: Learning generalizable semantic segmentation from sim-2-real with multi-resolution feature perturbation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.867104Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:2629da88b690a9450f7c281e97e56d823442e325724c4edad326d056d5848463","observation_id":"421215af-8ad4-40dc-b6ac-f8264a09775b","resolution":{"observed_at":"2026-08-07T11:00:54.867104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.871492Z","title":"Robust fine-tuning of zero-shot models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.871492Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:791c641e6aebda0b3ccb51ae4b58b7a388a99d28bd8dfcb8bebe38bc72bdf8f6","observation_id":"b2fb9c41-8b64-4f50-8232-ff423100053e","resolution":{"observed_at":"2026-08-07T11:00:54.871492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.875749Z","title":"Open-vocabulary panop- tic segmentation with text-to-image diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.875749Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:6e442b6891e183f48f86a7d54143ef35bab394f49035c5e9ee542df63752ddc4","observation_id":"733ca119-7afd-4c32-ab4b-5ca1414260bd","resolution":{"observed_at":"2026-08-07T11:00:54.875749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.127274Z","title":"A simple baseline for open- vocabulary semantic segmentation with pre-trained vision- language model","venue":null,"work_id":"cfaf407d-0dcd-4b1a-a952-9de27d2f2586","year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.880064Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:c16f34f110b8d2871693dbb5eaa8b2341ab82219204e79369d3de2bb6303422a","observation_id":"d3158162-599b-40c0-bdc4-3f476c4dc7bd","resolution":{"observed_at":"2026-08-07T11:00:55.131905Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.111447Z","title":"Side adapter network for open-vocabulary semantic segmentation","venue":null,"work_id":"2242bb49-0541-49ba-a9d4-0c38279b9d34","year":2023},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.884328Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:58b3c4bf42f8f6362d226ef4da75e300c054697dcf98e7f2a93bf8ad60da493f","observation_id":"2e0849b6-a601-4887-a56d-6f4e2d2eff9e","resolution":{"observed_at":"2026-08-07T11:00:55.116220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.888526Z","title":"Convolutions die hard: Open-vocabulary seg- mentation with single frozen convolutional clip","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.888526Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:6b3217e63c504f4c60a768674729d3e489813cd25b85a24e9376bef98ee2b19e","observation_id":"a81cd960-f265-4d4f-b8ae-2be8a902effd","resolution":{"observed_at":"2026-08-07T11:00:54.888526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.11432","last_updated":"2021-11-22T18:59:55Z","snapshot_observed_at":"2026-07-06T12:11:02.119174Z","submitted_at":"2021-11-22T18:59:55Z","title":"Florence: A New Foundation Model for Computer Vision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.11432","snapshot_observed_at":"2026-08-07T11:00:54.892839Z","title":"Florence: A new foundation model for computer vision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.892839Z"},"links":{"cited_paper":"/paper/2111.11432","citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:9cc7b20be1494efe44e47c43a722398b73b7f853424190bb5a3c764ff075ee5e","observation_id":"6ccf68ea-ae7e-4348-8e1b-26ffb316dae1","resolution":{"observed_at":"2026-08-07T11:00:54.892839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.080456Z","title":"Dept: Decoupled prompt tuning","venue":null,"work_id":"b0c4dfba-cb58-43ee-a6e6-b7e58d3ecc5b","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.897416Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:24eb79b032456ed1679f6da017e61a349afeeb866a56c14bf0888520656873d4","observation_id":"f4e0a183-1163-4ecf-8f73-f270a7e94a20","resolution":{"observed_at":"2026-08-07T11:00:55.085451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.064514Z","title":"Semantic under- standing of scenes through the ade20k dataset","venue":null,"work_id":"4236ba84-bbff-4276-81d0-88ab48cc60b4","year":2019},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.901790Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:766545c9f625c63ddd33f10d4f327a069a591245bbcc6b23be865732791c7930","observation_id":"d905bb80-3118-487a-bd14-c44d85bafd18","resolution":{"observed_at":"2026-08-07T11:00:55.069426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.906244Z","title":"Extract free dense labels from clip","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.906244Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:20d57732c300a36615e9e7066067a71a91716b96fd72441c162ab48134b4b2d1","observation_id":"0738a1bc-72a2-489a-8a01-338c1c816f09","resolution":{"observed_at":"2026-08-07T11:00:54.906244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.910555Z","title":"Generalized decoding for pixel, image, and lan- guage","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.910555Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:5984d5741439a1c6cb4f574da9dcd24c55d0b4c5b7e88964a4edc02e74cd9c2e","observation_id":"c03845bb-6da7-42aa-9abe-26da37c6d0dc","resolution":{"observed_at":"2026-08-07T11:00:54.910555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T11:23:24.568374Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":15,"verified_exact":1,"verified_fuzzy":22},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 0 inbound Pith citation observations for arXiv:2506.03709."}