{"as_of":"2026-08-11T12:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3623f8fd5d782c51aa55da052718ea7f7b0bd731d7e50feafe51a1843c56ad21","coverage":[{"denominator":83,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":83,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T19:48:36.016883Z","state":"measured"},{"denominator":83,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":83,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.09688/citation-record","integrity":"/paper/2501.09688/integrity","json":"/paper/2501.09688/citation-record.json","paper":"/paper/2501.09688"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:34.551388Z","title":"Label-embedding for image classification","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.551388Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:66996c4de5e394537ac9f8c665d38ff9dd4c220e3f4d7efba3ce262554a5be9e","observation_id":"090ca234-56dd-44bb-9fbb-8c4ed0a49cd8","resolution":{"observed_at":"2026-08-10T19:48:34.551388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:34.556685Z","title":"Evaluation of output embeddings for fine-grained image classification","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.556685Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:58c4c27871037d34bdc08dfa970de014dad20100d46c4b39a3984e32996d6b88","observation_id":"3e10791d-adac-4413-a546-65a5294568ef","resolution":{"observed_at":"2026-08-10T19:48:34.556685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:34.561748Z","title":"Paco: a novel procrustes application to co- phylogenetic analysis","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.561748Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:ce4945d77d437ac10430e183683a59d124edadc16a1e7579b666f26f5bc20198","observation_id":"2f808f93-d0aa-44e9-acd8-285a5bb19532","resolution":{"observed_at":"2026-08-10T19:48:34.561748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:39.490216Z","title":"Zero-shot semantic segmentation.Advances in Neural Information Processing Systems, 32, 2019","venue":null,"work_id":"1cac7ea5-b07b-4538-ba98-835025f3c9cd","year":2019},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.567215Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:7c50b0c32bf4e2b8bfb1dd8d831ef9e9fb45c070a8fdac4027a31185e9ccfd5d","observation_id":"6cbbd2d7-74a0-401f-b102-f9b4038c525f","resolution":{"observed_at":"2026-08-10T19:48:39.564928Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:39.475131Z","title":"Emerg- ing properties in self-supervised vision transformers","venue":null,"work_id":"b9f85eee-c373-4874-839e-6b2edb8d2f9b","year":2021},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.571723Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:b30ce2c5df037883a96a5028666b7f62fe6865ca0b97de5f0aa00b9a7c17c664","observation_id":"65078026-410d-4c78-a478-67f154958cfa","resolution":{"observed_at":"2026-08-10T19:48:39.480034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1706.05587","last_updated":"2017-12-05T18:06:21Z","snapshot_observed_at":"2026-08-07T13:44:53.690521Z","submitted_at":"2017-06-17T22:48:57Z","title":"Rethinking Atrous Convolution for Semantic Image Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.05587","snapshot_observed_at":"2026-08-10T19:48:34.576387Z","title":"Rethinking atrous convolution for semantic image segmentation","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.576387Z"},"links":{"cited_paper":"/paper/1706.05587","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:f2b0a3893404104215ebcb7c4eddfb39822c08335b0d1d24a412cf88039f0888","observation_id":"e5294c37-e609-4e46-8a3c-cf8095d8f3e0","resolution":{"observed_at":"2026-08-10T19:48:34.576387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10320","last_updated":"2023-05-17T16:01:27Z","snapshot_observed_at":"2026-08-10T21:26:41.711794Z","submitted_at":"2023-05-17T16:01:27Z","title":"CostFormer:Cost Transformer for Cost Aggregation in Multi-view Stereo","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.10320","snapshot_observed_at":"2026-08-10T19:48:34.582607Z","title":"Costformer: Cost transformer for cost aggregation in multi-view stereo","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.582607Z"},"links":{"cited_paper":"/paper/2305.10320","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:df7f2eebdb6cea532a02aac6f5e3602dba1ef65359f11aad3041186ef055a69a","observation_id":"85f6eeff-ba27-4a71-b4bd-8cadb962c36b","resolution":{"observed_at":"2026-08-10T19:48:34.582607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:34.587914Z","title":"Detect what you can: Detecting and representing objects using holistic mod- els and body parts","venue":null,"work_id":null,"year":1971},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.587914Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:362eba31f40c8434d6ba441e4939e588d5e4bc4734eabb92f9c06b94a73c206f","observation_id":"b16815cb-f87c-422f-9d19-db49140c377a","resolution":{"observed_at":"2026-08-10T19:48:34.587914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:39.449024Z","title":"Per- pixel classification is not all you need for semantic segmen- tation","venue":null,"work_id":"f6367afc-0dd2-412d-b519-6c5e6a864f75","year":2021},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.593114Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:8875f97ae13dd0911325f929d061815e5d12c7faa472f7632cb64fdd454718a9","observation_id":"960bfbdc-2388-4ebf-b73c-ae8aff1459b1","resolution":{"observed_at":"2026-08-10T19:48:39.454214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:39.391794Z","title":"Masked-attention mask transformer for universal image segmentation","venue":null,"work_id":"13589fa9-8659-477d-bf75-ea17dd31c5ee","year":2022},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.597669Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:f99fc99d585976cf0813bd7df700f6277161ba9fe548c1f0eb0041c7fc804d4f","observation_id":"017410d7-a138-4a38-a0e6-e682bb6ec7a4","resolution":{"observed_at":"2026-08-10T19:48:39.420381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:39.242370Z","title":"Cats: Cost aggre- gation transformers for visual correspondence","venue":null,"work_id":"e218dc90-9f0a-41c5-90ca-1d586f2f253f","year":null},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.602289Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:04e1893c99c9fa2e31d0614d773e9463b40a06358af8c82624c9d2104475e308","observation_id":"3c8c3fda-25fd-4659-92ca-def64e05f723","resolution":{"observed_at":"2026-08-10T19:48:39.318158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:39.101212Z","title":"Cats++: Boosting cost aggregation with convolutions and transformers","venue":null,"work_id":"0095e2ec-a2df-42c9-bd40-42d2cb37c0de","year":2022},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.606902Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:a9926bb2329faf915c80cfaec76d84f2b243eba2260afe6ce151d32b4a74f544","observation_id":"42b2915f-ac15-49a0-b8f1-90eabb1b2121","resolution":{"observed_at":"2026-08-10T19:48:39.156086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.11797","last_updated":"2024-03-31T11:53:55Z","snapshot_observed_at":"2026-08-10T15:40:12.549701Z","submitted_at":"2023-03-21T12:28:21Z","title":"CAT-Seg: Cost Aggregation for Open-Vocabulary Semantic Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.11797","snapshot_observed_at":"2026-08-10T19:48:34.611070Z","title":"Cat-seg: Cost aggregation for open-vocabulary semantic segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.611070Z"},"links":{"cited_paper":"/paper/2303.11797","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:296e23cf7d4c708b8033d6e2df371130dff47c25ec39f20aaf27c292a829abd3","observation_id":"258a8713-8db5-4b25-a073-76d7ab53602c","resolution":{"observed_at":"2026-08-10T19:48:34.611070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11384","last_updated":"2024-12-06T17:26:27Z","snapshot_observed_at":"2026-07-06T18:32:06.826381Z","submitted_at":"2024-06-17T10:11:28Z","title":"Understanding Multi-Granularity for Open-Vocabulary Part Segmentation","version":3},"cited_work":{"arxiv_id":"2406.11384","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.11384","snapshot_observed_at":"2026-08-10T19:48:36.321784Z","title":"Understanding Multi-Granularity for Open-Vocabulary Part Segmentation","venue":"cs.CV","work_id":"5c95e2a9-d870-4d36-ae96-9cdc63db54a0","year":2024},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.615802Z"},"links":{"cited_paper":"/paper/2406.11384","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:3bda4e8c0a513b8b217971db1105fb3beb65d9498e9da0089424f84740445b61","observation_id":"090ff2d1-46c1-453f-81ae-81c11ef56891","resolution":{"observed_at":"2026-08-10T19:48:36.359149Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:39.085261Z","title":"Unsupervised part discovery from con- trastive reconstruction","venue":null,"work_id":"3d4c9d14-db1c-4e7a-bbb5-291ed23ffd06","year":2021},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.620404Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:dc5eb363eab5e3ebf7a306fccd3a92c0c0c4d8021cc42d176a9e649c07247a7a","observation_id":"c80a8c29-2b16-4309-95c7-f85850fd7924","resolution":{"observed_at":"2026-08-10T19:48:39.090221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:34.625031Z","title":"Histograms of oriented gra- dients for human detection","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.625031Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:78f47aedb3af190cd0f68e1a13734cf3aa4958841c1216a0f801937269f278a5","observation_id":"a2df2f78-3101-438a-815d-8fc89ea04d1a","resolution":{"observed_at":"2026-08-10T19:48:34.625031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:39.055820Z","title":"Imagenet: A large-scale hierarchical image database","venue":null,"work_id":"7fd044f5-97ad-412d-b38e-07fd9a748408","year":2009},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.666573Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:e60d3bb4f41bed00510c187a258b2720c3520cc5cbf849a390cbee865f97050d","observation_id":"a8d0c380-4cc9-49ff-83eb-8a7e575ce328","resolution":{"observed_at":"2026-08-10T19:48:39.060716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2208.08984","last_updated":"2023-06-08T06:35:33Z","snapshot_observed_at":"2026-07-06T13:43:13.386359Z","submitted_at":"2022-08-18T17:55:37Z","title":"Open-Vocabulary Universal Image Segmentation with MaskCLIP","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2208.08984","snapshot_observed_at":"2026-08-10T19:48:34.697077Z","title":"Open- vocabulary universal image segmentation with maskclip","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.697077Z"},"links":{"cited_paper":"/paper/2208.08984","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:70d53043c65610f2b7cce288980bc338b274f9db2c5cbdaed37f40097ae7866b","observation_id":"2c478850-7782-4a0c-b1d1-dc5066260f8b","resolution":{"observed_at":"2026-08-10T19:48:34.697077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:39.038839Z","title":"Generalized jensen- shannon divergence loss for learning with noisy labels","venue":null,"work_id":"812ae4b1-5e4e-42d7-a8a3-f11b81b04fc1","year":2021},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.726678Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:9e7ff7dbe2caee426b7bba04cb627dab042f39bdd9cf812106c27eeaba537c34","observation_id":"4dc9a23b-7481-4ac6-af53-ceb46f007112","resolution":{"observed_at":"2026-08-10T19:48:39.044496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.930807Z","title":"De- vise: A deep visual-semantic embedding model","venue":null,"work_id":"3b1a7a89-49dd-4731-a65d-9d6e02552557","year":2013},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.746450Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:0e5b270a773b1a207169b67e25750b31afcc49ecabd44454c9ee000253c91d70","observation_id":"06cadc3c-d79f-4da8-a2ef-29f2059df66a","resolution":{"observed_at":"2026-08-10T19:48:39.020580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.914035Z","title":"Scal- ing open-vocabulary image segmentation with image-level labels","venue":null,"work_id":"d4f976fb-6d9c-4776-bbd3-a1754cb0f27e","year":2022},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.769817Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:0fb67945dc0ce9a84e112b4af209e476f0ca3cfaa2b9568530221e8416667610","observation_id":"fc469418-4c88-4de4-987b-31bace6db67c","resolution":{"observed_at":"2026-08-10T19:48:38.919498Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.13921","last_updated":"2022-05-12T01:27:40Z","snapshot_observed_at":"2026-08-08T10:08:57.896975Z","submitted_at":"2021-04-28T17:58:57Z","title":"Open-vocabulary Object Detection via Vision and Language Knowledge Distillation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.13921","snapshot_observed_at":"2026-08-10T19:48:34.792236Z","title":"Open- vocabulary object detection via vision and language knowl- edge distillation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.792236Z"},"links":{"cited_paper":"/paper/2104.13921","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:141f8b26007270807b6e403d309f4fa50583db190d88d2eb82b074ae7605394a","observation_id":"6235dc49-b571-43c3-a57b-ae6e211aa201","resolution":{"observed_at":"2026-08-10T19:48:34.792236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.897342Z","title":"Aˆ 3: Accelerating attention mechanisms in neural networks with approximation","venue":null,"work_id":"426a4a82-8479-4d5c-a716-70eb002b85e0","year":2020},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.815372Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:8286e279385bee8c9fe9a14554e22b4e4276f90e994b92039ff1624c8b290c7d","observation_id":"fdd10945-a73b-4b98-ac2b-1ef077c1edbf","resolution":{"observed_at":"2026-08-10T19:48:38.903395Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.881018Z","title":"Global knowledge calibration for fast open-vocabulary segmentation","venue":null,"work_id":"c7d770b3-f117-4bb2-b938-fb98ea11a1ac","year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.839230Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:74a1398e23aac1b4a8053ac4c20ed41b417da7662f09bf0653839705a3dfbce2","observation_id":"87b5c4ee-a96b-4463-a708-630252f86d16","resolution":{"observed_at":"2026-08-10T19:48:38.886589Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.835159Z","title":"Partimagenet: A large, high- quality dataset of parts","venue":null,"work_id":"6f465df8-695c-4da6-9017-423eda603f11","year":2022},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.845068Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:5fd544a363eb6c32ab0f4822241a64bff7c47bc9d6f7f8e0e458982bf999e1b6","observation_id":"a27ce1e4-aef7-4885-b14d-15ea7e063f22","resolution":{"observed_at":"2026-08-10T19:48:38.869563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.767034Z","title":"Compositor: Bottom-up clustering and compositing for robust part and object segmentation","venue":null,"work_id":"4dada973-5716-4a8f-945f-9e49de5c4be3","year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.853836Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:bb714e09b1737e952cfb0701c83ff97e9e53dcfb2fd3470ab9ead1321b5f189c","observation_id":"36c60660-742d-46e9-a6cc-d3ca80b2c58b","resolution":{"observed_at":"2026-08-10T19:48:38.805186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:34.865219Z","title":"Deep residual learning for image recognition","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.865219Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:b1892e261e15895afa38c2264a441eb1aee099684e08d9ee188f9fd4aaa6d150","observation_id":"70df0c52-92af-403e-863d-ad6011ba1586","resolution":{"observed_at":"2026-08-10T19:48:34.865219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.637306Z","title":"Mask r-cnn","venue":null,"work_id":"4c0a8a4d-2ab3-4271-ba16-d8925de3b9cc","year":2017},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.892492Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:941d4c4255f843ec728785079094c1177a12eb46fd2b31d78153b3e30a07a8ef","observation_id":"18c90947-ced4-4015-9b1a-bebec4c8ff41","resolution":{"observed_at":"2026-08-10T19:48:38.682497Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.621807Z","title":"Cost aggregation with 4d convolutional swin transformer for few-shot segmentation","venue":null,"work_id":"6670e89c-5d13-414b-83ac-82a4c6091c85","year":null},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.906538Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:6ee52f420b87c5534cc8fbefd16f5d41dc24754fac0a069d1b34df1ccdbd6d96","observation_id":"7daf053a-b5a3-4517-8205-b2b6b1338d37","resolution":{"observed_at":"2026-08-10T19:48:38.626851Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.605105Z","title":"Unifying feature and cost aggregation with transformers for semantic and visual correspondence","venue":null,"work_id":"b58fbb23-a4ce-4182-ac77-7dda32b9e9d2","year":2024},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.931238Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:f8ed0022a4db03b8c4c1af6ef31e82109752592edbabd0c7576b15679c1324fb","observation_id":"3d87664e-c3e1-4449-809e-564d669b56c6","resolution":{"observed_at":"2026-08-10T19:48:38.610800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.589232Z","title":"Fast cost-volume filtering for visual correspondence and beyond","venue":null,"work_id":"6aceeee6-bd17-4b0f-8542-86b9d0d79f3a","year":null},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.949337Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:2751bbbed6730cc47d98f37f2f6b1b87bddcf84cdf8bb08ab49007b64c9f451d","observation_id":"f83b5638-299d-49c2-a4a6-d65f736e96ca","resolution":{"observed_at":"2026-08-10T19:48:38.594462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:34.956760Z","title":"Scops: Self-supervised co-part segmentation","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.956760Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:170a7eba63c68db0f491d76997eb8270fce128e351b0c8dce23094b20e12d41c","observation_id":"c9c64778-b3ea-45c8-8585-60d622656e39","resolution":{"observed_at":"2026-08-10T19:48:34.956760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:34.968102Z","title":"Scaling up visual and vision-language representa- tion learning with noisy text supervision","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.968102Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:d9f5da4b1198dbc19be2c8a18dfd05cb3cf7f2b62c11d7abe37bcb5528b70683","observation_id":"b46d92f2-81db-469c-bf86-f67cab1766e1","resolution":{"observed_at":"2026-08-10T19:48:34.968102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.488563Z","title":"Salad: Part-level latent diffusion for 3d shape gen- eration and manipulation","venue":null,"work_id":"313031d0-18d1-4e74-b02e-a25012eb45c3","year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.973831Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:3902718e7d7980389cb00365f2b58a8b62ade5a69bf5e4152e8046880f5220cf","observation_id":"2111dec3-a319-4d3a-bfc2-e79983edec3d","resolution":{"observed_at":"2026-08-10T19:48:38.526142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.03546","last_updated":"2022-04-03T03:33:43Z","snapshot_observed_at":"2026-08-01T06:14:31.660540Z","submitted_at":"2022-01-10T18:59:10Z","title":"Language-driven Semantic Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.03546","snapshot_observed_at":"2026-08-10T19:48:34.980861Z","title":"Language-driven semantic seg- mentation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.980861Z"},"links":{"cited_paper":"/paper/2201.03546","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:e1508d1625091c86bcbe02fb9a849be4aff2bd6dc61ccb2bb9de2467c28c9765","observation_id":"fa1371b3-37bf-49f7-8c91-1d4948253fcc","resolution":{"observed_at":"2026-08-10T19:48:34.980861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.382758Z","title":"Mask dino: Towards a unified transformer-based framework for object detection and segmentation","venue":null,"work_id":"9bfcedeb-6a7c-4664-bfef-6aba3e5c295b","year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.986851Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:3e1a83b365e7b5a24786d9b0d6cb806264ef0677fab3365dd62a7e897ed1e42b","observation_id":"dca016f8-153a-4fcf-b9dc-9a88c5d6a78a","resolution":{"observed_at":"2026-08-10T19:48:38.427123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.365972Z","title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","venue":null,"work_id":"340ef2fa-d980-42f0-b698-ac270acc534e","year":2022},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.992238Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:f9881ce7d5812e352bb3e600164bb4ecab11158e4e1ea7f41ea7653e92db8aa5","observation_id":"84d41189-f081-49c7-894d-9a08336ca5af","resolution":{"observed_at":"2026-08-10T19:48:38.371041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16696","last_updated":"2024-07-23T17:58:26Z","snapshot_observed_at":"2026-07-06T18:50:52.800619Z","submitted_at":"2024-07-23T17:58:26Z","title":"PartGLEE: A Foundation Model for Recognizing and Parsing Any Objects","version":1},"cited_work":{"arxiv_id":"2407.16696","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.16696","snapshot_observed_at":"2026-08-10T19:48:36.235340Z","title":"PartGLEE: A Foundation Model for Recognizing and Parsing Any Objects","venue":"cs.CV","work_id":"1fa7fe4e-5aff-46ce-9f7a-434c53e14711","year":2024},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:34.997403Z"},"links":{"cited_paper":"/paper/2407.16696","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:95523752c896d036ab88d9edfc6281003d211ff31a7dc93c6351845595e269f7","observation_id":"412c3d69-cee2-4418-aca0-8ab265359378","resolution":{"observed_at":"2026-08-10T19:48:36.240407Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05653","last_updated":"2024-09-16T09:10:00Z","snapshot_observed_at":"2026-08-10T13:07:09.026586Z","submitted_at":"2023-04-12T07:16:55Z","title":"A Closer Look at the Explainability of Contrastive Language-Image Pre-training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05653","snapshot_observed_at":"2026-08-10T19:48:35.003827Z","title":"Clip surgery for better explainability with enhancement in open- vocabulary tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.003827Z"},"links":{"cited_paper":"/paper/2304.05653","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:9d87d503edc8fe93aeb8fd3d23e79a95a4abbb19f875d18bb6c7952df7186497","observation_id":"db745552-0884-4e75-b567-b48044445785","resolution":{"observed_at":"2026-08-10T19:48:35.003827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:35.009952Z","title":"Open-vocabulary semantic segmentation with mask-adapted clip","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.009952Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:7ffd37f74437f555d5fd46c1d3e0248002e7c40cd4ef19a1c05744b8ff7abc7d","observation_id":"fd663b1c-fa4d-4d12-a5b1-c3354e95722d","resolution":{"observed_at":"2026-08-10T19:48:35.009952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:35.014541Z","title":"Divergence measures based on the shannon en- tropy","venue":null,"work_id":null,"year":1991},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.014541Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:493735955f283f9b77081e143ac0a6b3c319cf46b714ae16fa048f6cf4f503d4","observation_id":"211d525b-f2e8-46a8-9a14-e31c3955d583","resolution":{"observed_at":"2026-08-10T19:48:35.014541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.331613Z","title":"Editgan: High-precision semantic image editing","venue":null,"work_id":"b6f633e9-7ae0-4f00-9f09-92bcd50fb985","year":2021},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.019911Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:3d0114708b78ef1088147268a82478c8190347b27d1ee624476ffa6ae82430aa","observation_id":"3fdcd5fb-6d99-41d8-8102-35d80a7dd0ed","resolution":{"observed_at":"2026-08-10T19:48:38.336701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:35.034819Z","title":"Sift flow: Dense correspondence across scenes and its applications","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.034819Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:52c974dd1b75ac0bdf01699ed313c68e0023da43d2117ab2ea321cddaf19d9ff","observation_id":"43330f00-7824-4ae3-80a7-b36bf3d4c7fa","resolution":{"observed_at":"2026-08-10T19:48:35.034819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.305936Z","title":"Se- mantic correspondence as an optimal transport problem","venue":null,"work_id":"00d753b7-4534-44ed-82d5-b764ad0feb27","year":2020},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.040458Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:9a5911b3178684b6abb4fca76e54087e0ad7bd4d7a4fb24690ec30ca71fe4d91","observation_id":"ed449e01-2df5-4761-b57c-dbbc53ed5424","resolution":{"observed_at":"2026-08-10T19:48:38.311124Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04089","last_updated":"2024-11-26T13:45:09Z","snapshot_observed_at":"2026-07-06T16:58:11.501328Z","submitted_at":"2023-12-07T07:00:09Z","title":"Open-Vocabulary Segmentation with Semantic-Assisted Calibration","version":2},"cited_work":{"arxiv_id":"2312.04089","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.04089","snapshot_observed_at":"2026-08-10T19:48:36.136903Z","title":"Open-Vocabulary Segmentation with Semantic-Assisted Calibration","venue":"cs.CV","work_id":"e73b6c21-3881-462d-9e3b-0249c1afe615","year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.090187Z"},"links":{"cited_paper":"/paper/2312.04089","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:2bc3ebbb6e95fa3444ddf0250fdc8ae3a55af249bb95e39fddebd6c67d5cdb87","observation_id":"202b5526-63d1-4c6e-8ce2-e1a74fc9d079","resolution":{"observed_at":"2026-08-10T19:48:36.201592Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.290801Z","title":"3d part guided image editing for fine-grained object understanding","venue":null,"work_id":"aee557ed-3e42-43bf-b2ea-e18d96665761","year":2020},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.168028Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:af5d01e7aec6b0b723bb77b07b108ea0d1abdbbcda2d84aaea605d8309bcdf20","observation_id":"611ca277-4f8d-436c-a108-cde8ad692068","resolution":{"observed_at":"2026-08-10T19:48:38.295859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.222601Z","title":"Swin transformer: Hierarchical vision transformer using shifted windows","venue":null,"work_id":"4f3202cf-7de8-41a9-8578-a7a0cbf1163d","year":2021},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.220364Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:b2bd90f2409ebfc7321ce2dbdaed8aedab0ab57c0fc9839c3a81015e908ac208","observation_id":"86fd6094-08d6-4e51-943d-c608a4f07b0c","resolution":{"observed_at":"2026-08-10T19:48:38.277161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.061794Z","title":"Fully convolutional networks for semantic segmentation","venue":null,"work_id":"3a87e033-a266-4a93-afb1-f1d96e401bea","year":2015},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.238221Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:18afd95514c136807bdd7753a3251162f7173956bd379d69deafb2cc83616ab2","observation_id":"7f0f0b9c-914f-44b0-92ac-9df8af4805c5","resolution":{"observed_at":"2026-08-10T19:48:38.148560Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-09T20:34:52.923500Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-10T19:48:35.270590Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.270590Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:13e7a18365a6afce13e7684285d65bb4c9b250192373171eb36b8b1a391541d0","observation_id":"66bdb88f-8352-4ee8-a5aa-949b25c137ce","resolution":{"observed_at":"2026-08-10T19:48:35.270590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.046413Z","title":"Image segmenta- tion using text and image prompts","venue":null,"work_id":"8b891d7d-696e-4412-a981-4187e7dbea48","year":2022},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.284733Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:5d4f4ee490849752d5040fd7ae806bebf870aa86e1a1a0064a628f483919a2c2","observation_id":"f19caff0-bc32-4b04-a66a-11d598250310","resolution":{"observed_at":"2026-08-10T19:48:38.051535Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.030926Z","title":"Wordnet: a lexical database for english","venue":null,"work_id":"5c6b070d-ca83-42a8-a1fa-16c8df985a39","year":1995},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.290237Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:b5e912d2b1ef68747b7625c80575ebc64c7990af7119b551fed15a9efffbf0ae","observation_id":"619768ee-e4cb-4441-a17f-6cdbaa304d53","resolution":{"observed_at":"2026-08-10T19:48:38.036443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:38.014621Z","title":"Hyperpixel flow: Semantic correspondence with multi-layer neural features","venue":null,"work_id":"d6555fd1-81b5-44e2-b44d-369e095fd042","year":null},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.295258Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:6722c593e3fa6cf305f8221b9d654e08152fb0ff0ce203b1baf8fb1c20e41369","observation_id":"7e2e7a7b-f640-4ab1-a2b8-119db16410c2","resolution":{"observed_at":"2026-08-10T19:48:38.019994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:37.975792Z","title":"Learning to compose hypercolumns for visual correspon- dence","venue":null,"work_id":"94f0bb7e-8323-4394-a394-b9cde2bd23fe","year":2020},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.300333Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:d0ca4471a34c344f99f782a68616cb35f9c6c72890699c3b7fbbcb4e58fed431","observation_id":"80e081cf-dc03-4f6f-bbbe-9f18dc69bb87","resolution":{"observed_at":"2026-08-10T19:48:37.992362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-08-11T10:12:11.384939Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-10T19:48:35.305078Z","title":"Dinov2: Learning robust visual features without supervision","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.305078Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:15e12cbd9849f87fed9485895a4583cde863c48c5a94878d9ddf729e4cb08f1b","observation_id":"0e219efb-c350-4ad7-bafa-d0f6cdb3e00d","resolution":{"observed_at":"2026-08-10T19:48:35.305078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:37.917873Z","title":"To- wards open-world segmentation of parts","venue":null,"work_id":"0ce80626-77c1-452b-987e-98235f0592bf","year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.309881Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:d8b32f831138f502e6d9b0367d6e50ceb7d8b17e46afa703a459259950f96675","observation_id":"8b02dd13-5452-43aa-b523-22bd10a27920","resolution":{"observed_at":"2026-08-10T19:48:37.952351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:37.743327Z","title":"Computational optimal transport: With applications to data science.Foundations and Trends® in Machine Learning, 11(5-6):355–607, 2019","venue":null,"work_id":"ba26cedc-4b6d-44fc-80ba-e978a1d9a56f","year":2019},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.314296Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:034b510e4ba023095c0c99aa23df7531094b2dd38d6c267f8de820c85f93ee73","observation_id":"a1d534f8-1caf-4686-b0c4-6bda92071e15","resolution":{"observed_at":"2026-08-10T19:48:37.862486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:37.705309Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":"baae9487-9c43-42bf-855b-25f190661029","year":2021},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.318598Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:5f7827e05a6be4dbe26a4a8985799b8b2c28575ea2258a5f4b3ce4a2305e6372","observation_id":"b6521947-2918-498f-92ae-d0dbaac1feb9","resolution":{"observed_at":"2026-08-10T19:48:37.711139Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:37.689337Z","title":"Neighbourhood con- sensus networks","venue":null,"work_id":"a6706f96-5ea9-404a-9d9b-8d93069ac80c","year":2018},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.402164Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:6f790084bdbb017a7f5437a68aaae6d2348eb344c805ee6162f315ae2f9d8b45","observation_id":"6d18e1a8-32d5-4ff2-a73b-61630d19d78e","resolution":{"observed_at":"2026-08-10T19:48:37.695243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:35.475790Z","title":"U- net: Convolutional networks for biomedical image segmen- tation","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.475790Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:fc191ee5aaefca9a5fdaa196d1a3641e23e9f5d3169c57715567cf75e1ce5992","observation_id":"6e7561bd-6317-4b1a-a42c-ee91e747402c","resolution":{"observed_at":"2026-08-10T19:48:35.475790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:35.521113Z","title":"Pwc-net: Cnns for optical flow using pyramid, warping, and cost volume","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.521113Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:85c6991b7202c2457b3fe14d1324f3016958b37709f39924ab678fb34a1a151d","observation_id":"59452721-16ab-4ab2-aeab-4973fa80a5ce","resolution":{"observed_at":"2026-08-10T19:48:35.521113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:37.519472Z","title":"Going denser with open-vocabulary part segmentation","venue":null,"work_id":"151f029a-8968-4f77-919e-df9868f4bd49","year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.528808Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:13327cdcab8411a21fce224cdb34339c679f2b00320a8bdec6a62a3604dbe520","observation_id":"caaf3bd3-3989-44d0-a44b-fdc2a81be1ff","resolution":{"observed_at":"2026-08-10T19:48:37.603310Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:37.334988Z","title":"Parts and wholes in face recognition","venue":null,"work_id":"130abe54-6c75-4e7b-a818-c9d598322e94","year":1993},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.548344Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:343372afc3a3313322ceaf8858ba233b4d1b5596dc34866edf134917a1d41b77","observation_id":"8f782bcd-1d60-44cd-8084-b5685af8d568","resolution":{"observed_at":"2026-08-10T19:48:37.407465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:35.585172Z","title":"Glu- net: Global-local universal network for dense flow and corre- spondences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.585172Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:195d5ccef64c89c64beb666f9c8d48f215fff263a97d791a902d06a3831daa81","observation_id":"4a98ab58-959a-4065-a6e0-c904a9e51181","resolution":{"observed_at":"2026-08-10T19:48:35.585172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:35.643066Z","title":"Pdisconet: Semantically consistent part discovery for fine-grained recognition","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.643066Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:a20466e7336ff6ed72c605eaa647c045ff1be3a01a519cff5523ef070c86e865","observation_id":"6cbf97a8-48c1-45a1-a3f1-739676d60de5","resolution":{"observed_at":"2026-08-10T19:48:35.643066Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:35.670739Z","title":"Attention is all you need","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.670739Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:5ceb926e574d5808fa9cde68b6e28dae2c20a4005bbfaa0c4766f95535b6cbcb","observation_id":"3725fade-665e-4dcf-b2d1-5b39d852d46c","resolution":{"observed_at":"2026-08-10T19:48:35.670739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:37.289129Z","title":"Optimal transport: old and new","venue":null,"work_id":"b689892b-7b64-424a-9b43-b79a0375bf5a","year":2009},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.675236Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:5626911beb7d2c4ed191b0acad7ff50bc9682142e1393a4386771d028e689cb7","observation_id":"8ad3e1a2-f5b4-4b3d-9645-349a71d6a855","resolution":{"observed_at":"2026-08-10T19:48:37.294110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:37.273040Z","title":"In- structpart: Affordance-based part segmentation from lan- guage instruction","venue":null,"work_id":"5f0c0c74-e69a-43ff-aa5c-0e1116f0cdb0","year":2024},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.679273Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:0242fa11bd01eac6e33246a1e14150c2b4f875980cd4b17a1ca050a561653e3f","observation_id":"42bf180c-584e-4349-bb11-b2689d7bbf2b","resolution":{"observed_at":"2026-08-10T19:48:37.278144Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:37.246069Z","title":"Pyramid vision transformer: A versatile backbone for dense prediction without convolutions","venue":null,"work_id":"df5b2b33-d57a-4b80-b230-622fee66255d","year":2021},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.683582Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:742399d1b908a9f16df97ed476ad5ae4be2a15f2cd356bfc80881bc742012d77","observation_id":"baeead81-f078-4bc4-9c31-1eb858c9192e","resolution":{"observed_at":"2026-08-10T19:48:37.259446Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03290","last_updated":"2024-02-05T18:49:17Z","snapshot_observed_at":"2026-08-06T08:11:36.010824Z","submitted_at":"2024-02-05T18:49:17Z","title":"InstanceDiffusion: Instance-level Control for Image Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03290","snapshot_observed_at":"2026-08-10T19:48:35.687564Z","title":"Instancediffusion: Instance- level control for image generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.687564Z"},"links":{"cited_paper":"/paper/2402.03290","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:623ae4bd7551ab9307d3f5c1a161186b339d4424eed6d9fa2bb73cd01be7fbd9","observation_id":"356ad15a-7c21-48c1-90f3-fc090b51284f","resolution":{"observed_at":"2026-08-10T19:48:35.687564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:37.180052Z","title":"Ov-parts: Towards open- vocabulary part segmentation","venue":null,"work_id":"baee3eae-1603-46ae-9aa6-64bc56823c93","year":2024},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.691786Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:1b902c865e323b4c4a0cff2d338e554effbcc56e63529b8079fe05534605fff3","observation_id":"0c0411f4-e0f9-42f2-b1ac-38381ad30507","resolution":{"observed_at":"2026-08-10T19:48:37.227174Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:36.995560Z","title":"Semantic projection network for zero-and few-label semantic segmentation","venue":null,"work_id":"d68fc3bd-71f4-412a-be58-e2459c63090b","year":2019},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.787678Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:2a46f8582621206fe05e70d22bee55af928c2f8d9f9c8e7101244a2152e5eebb","observation_id":"da2551fc-7433-4889-bcd0-23e02b1e673b","resolution":{"observed_at":"2026-08-10T19:48:37.076600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15537","last_updated":"2024-02-27T10:59:30Z","snapshot_observed_at":"2026-07-06T16:52:53.954856Z","submitted_at":"2023-11-27T05:00:38Z","title":"SED: A Simple Encoder-Decoder for Open-Vocabulary Semantic Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.15537","snapshot_observed_at":"2026-08-10T19:48:35.818590Z","title":"Sed: A simple encoder-decoder for open-vocabulary semantic segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.818590Z"},"links":{"cited_paper":"/paper/2311.15537","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:76856417c1b975461b13de609eb1e5736079a188d66a29451f755c2aac7cd8d4","observation_id":"73d0a2c2-defd-4361-8643-d1ab7e9d7ffb","resolution":{"observed_at":"2026-08-10T19:48:35.818590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:35.892353Z","title":"Open-vocabulary panop- tic segmentation with text-to-image diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.892353Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:484479657ce92fc1fff22afca58c77b64c0ceb04f854955fddde130243fed2cc","observation_id":"e494ba32-11ca-430c-b152-c5ecf18e8f89","resolution":{"observed_at":"2026-08-10T19:48:35.892353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:36.894525Z","title":"A simple baseline for open- vocabulary semantic segmentation with pre-trained vision- 11 language model","venue":null,"work_id":"fbae483b-1cea-49e4-9bf4-a19eb67e1e24","year":2022},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.920269Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:b315bb01a3ba18376aeaac38a7dfbdda7d365a00177133c67e6f54166ad31370","observation_id":"7ad2a31a-3111-4423-a6d7-84bc4f46a561","resolution":{"observed_at":"2026-08-10T19:48:36.899779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.11565","last_updated":"2024-01-10T14:20:30Z","snapshot_observed_at":"2026-08-05T08:41:47.424828Z","submitted_at":"2023-06-20T14:30:32Z","title":"HomeRobot: Open-Vocabulary Mobile Manipulation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.11565","snapshot_observed_at":"2026-08-10T19:48:35.950172Z","title":"Homerobot: Open-vocabulary mobile manipulation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.950172Z"},"links":{"cited_paper":"/paper/2306.11565","citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:fbfa9328d1e8bbcb9932732867cbc5f2e9258bbc7a2ca3e597e4e33bf52ad608","observation_id":"02c90f8f-228b-48c9-ac0e-3cf840ace960","resolution":{"observed_at":"2026-08-10T19:48:35.950172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:36.877187Z","title":"C2fnas: Coarse- to-fine neural architecture search for 3d medical image seg- mentation","venue":null,"work_id":"e47ab772-c0ed-4067-b0b5-d37f47ef9028","year":2020},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.984489Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:b7cc68e59c79a5d8243e4f07e55e75428e5326540608e313667de27484788df0","observation_id":"8f091fff-f0d5-4d06-8878-ec5c753b7a37","resolution":{"observed_at":"2026-08-10T19:48:36.882710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:36.860743Z","title":"Convolutions die hard: Open-vocabulary seg- mentation with single frozen convolutional clip.Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":"115e68d1-32d3-4f41-8bb1-8e566769ad12","year":2024},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.989294Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:aa302dc7e3a4c024f714a9b5c348aa0d837e4ad1bf4b5fa2996d8189cc0c736f","observation_id":"bd5c994d-3bb6-4156-83c4-f5307238b284","resolution":{"observed_at":"2026-08-10T19:48:36.866331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:36.807534Z","title":"Open-vocabulary object detection using captions","venue":null,"work_id":"aefc63de-5b28-456d-8606-e47a37842a5f","year":2021},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.994288Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:7b1b175c5ece4f98a246e3ff97a8c4dbaec2239f54d6ea1ad9cc64b54fc6cad2","observation_id":"bfdf972c-adc4-4cb7-9833-a38f74b2cb8c","resolution":{"observed_at":"2026-08-10T19:48:36.824494Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:36.693585Z","title":"Open vocabulary scene parsing","venue":null,"work_id":"8a1f3dc1-bed4-49ca-8456-3cba4f015d98","year":2002},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:35.999110Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:f2e0b64b82859425c38492e7fb94aa4f20a8ca455f0e276ca06973062788091c","observation_id":"d9dca9e4-dc36-410a-acb5-14709fc7a54f","resolution":{"observed_at":"2026-08-10T19:48:36.729546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:36.003829Z","title":"Scene parsing through ade20k dataset","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:36.003829Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:98c6294b5b1835d8918bc0d4c4a5210462d8851c52c6a046a9f53b39da51f209","observation_id":"af7668bc-2420-4675-ac2a-d7ae68aca7a6","resolution":{"observed_at":"2026-08-10T19:48:36.003829Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:36.610046Z","title":"Extract free dense labels from clip","venue":null,"work_id":"7f06bce1-7053-4137-bf74-b8459c3e6782","year":2022},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:36.008198Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:7b109b3e2369ac872ef60cffa073960efff2f24de134a84af2d3a444765aded9","observation_id":"77201593-316d-4da1-b232-5debc540a1b0","resolution":{"observed_at":"2026-08-10T19:48:36.632220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:36.012707Z","title":"Learning to prompt for vision-language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:36.012707Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:07dc69a59825d35ce33d77206d974f7415bf2d238b64aff764246cae7ac2ef09","observation_id":"b5d453f8-a3e4-410f-81a9-fa32156e1a75","resolution":{"observed_at":"2026-08-10T19:48:36.012707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T19:48:36.487594Z","title":"person” (b) “person’s eye","venue":null,"work_id":"93c8b1a7-a342-407a-8ab9-0e92b899b55c","year":2023},"citing_paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-10T19:48:36.016883Z"},"links":{"citing_paper":"/paper/2501.09688"},"observation_digest":"sha256:653a367fa159032661505400adc941f06c27f81c079e6e9f20ae2d6d0f0731d0","observation_id":"788bf27f-901c-49ce-9303-eb0147d25a54","resolution":{"observed_at":"2026-08-10T19:48:36.543756Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.09688","last_updated":"2025-08-08T08:51:23Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-10T19:43:10.754562Z","submitted_at":"2025-01-16T17:40:19Z","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation"},"reference_resolution":{"displayed":83,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":31,"verified_exact":3,"verified_fuzzy":49},"total_outbound_references":83},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 83 of 83 outbound references and 0 inbound Pith citation observations for arXiv:2501.09688."}