{"as_of":"2026-08-17T05:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9bae52afd9a8a2746e692cbf446e14758a9314413e2cc9c1f48def4384651808","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T15:53:31.889301Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T20:04:44.106636Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-13T22:21:15.797283Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13836","snapshot_observed_at":"2026-08-12T20:04:44.106636Z","title":"Cliper: Hierarchically improving spatial represen- tation of clip for open-vocabulary semantic segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.10086","last_updated":"2025-08-01T08:25:34Z","snapshot_observed_at":"2026-08-16T21:18:00.617750Z","submitted_at":"2024-11-15T10:14:55Z","title":"CorrCLIP: Reconstructing Patch Correlations in CLIP for Open-Vocabulary Semantic Segmentation","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T20:04:44.106636Z"},"links":{"cited_paper":"/paper/2411.13836","citing_paper":"/paper/2411.10086"},"observation_digest":"sha256:be35a92335b01dfdbfd7a9bdde675c70356475f9519cf99d9da51210e8108823","observation_id":"a7d5046b-d83a-4772-9b7f-346ca2037fe4","resolution":{"observed_at":"2026-08-12T20:04:44.106636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"cited_work":{"arxiv_id":"2411.13836","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13836","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"CLIPer: Hierarchically improving spatial representation of CLIP for open-vocabulary semantic segmentation","venue":null,"work_id":"51b1928f-fa27-4f57-8134-9f4b0d2e5e33","year":2024},"citing_paper":{"arxiv_id":"2504.13181","last_updated":"2025-04-28T18:01:39Z","snapshot_observed_at":"2026-08-11T19:32:15.215968Z","submitted_at":"2025-04-17T17:59:57Z","title":"Perception Encoder: The best visual embeddings are not at the output of the network","version":2},"reference_index":128,"source":"pdf_text","source_observed_at":"2026-05-13T22:21:15.681336Z"},"links":{"cited_paper":"/paper/2411.13836","citing_paper":"/paper/2504.13181"},"observation_digest":"sha256:884990dc9bbbe261d5650fe6c50c6a6c9e868a1513ce8a6e5b9841cf724744de","observation_id":"1fee426a-3576-4ef6-b5f8-473170d116dc","resolution":{"observed_at":"2026-05-13T22:21:15.798787Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.13836/citation-record","integrity":"/paper/2411.13836/integrity","json":"/paper/2411.13836/citation-record.json","paper":"/paper/2411.13836"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:35.071758Z","title":"Weakly su- pervised learning of instance segmentation with inter-pixel relations","venue":null,"work_id":"5ec0f6b6-a556-408c-b6f0-9f3fe274eb0b","year":2019},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.953476Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:06311f860b9ce79a5f672828a59c171fa8e8475f33cdea5488905c6f82faf7d9","observation_id":"e8d4222c-c7a9-42dd-8f17-90c0bb26a74d","resolution":{"observed_at":"2026-08-12T15:53:35.104096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.967475Z","title":"Zero-shot semantic segmentation","venue":null,"work_id":"f3dc0727-f386-4ca5-a37d-a11c59d07569","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.960642Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:a22e4593feb079e2515c09938733601b387ee796e9d962b0665e4eda9bd070bd","observation_id":"6c6c5a6a-84a4-4045-aa37-4967a6e6d61d","resolution":{"observed_at":"2026-08-12T15:53:35.009572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.878577Z","title":"Coco- stuff: Thing and stuff classes in context","venue":null,"work_id":"ea67ebe2-0c84-4293-92dd-29e12dd7c464","year":2018},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.967487Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:be7ce87feb5232ebde0bc62249ca5e1d8f5d70b7ccaa1670d777586441d28c15","observation_id":"d18a48e3-a586-4cd5-b360-d17f0e54e0a9","resolution":{"observed_at":"2026-08-12T15:53:34.916255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.828422Z","title":"Emerg- ing properties in self-supervised vision transformers","venue":null,"work_id":"9698c3b4-1acb-4047-ab46-b92982c838ff","year":2021},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.973143Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:d74d377e94e53a4487126584be222123bcd37d8ef388af24f1baf3ae80ce4398","observation_id":"197c4d1a-34e0-4973-abff-865a2fdd9132","resolution":{"observed_at":"2026-08-12T15:53:34.852705Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.768051Z","title":"Learn- ing to generate text-grounded mask for open-world semantic segmentation from only image-text pairs","venue":null,"work_id":"4b36e1f9-e0c7-4f6a-9126-2990de5927e9","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.980180Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:e1cc6092586ab006504c5fdef6b8a4ec8959adf4da32047e387894ebc434ad61","observation_id":"2fb29648-8d04-4c9b-b931-1e7a56e978bd","resolution":{"observed_at":"2026-08-12T15:53:34.806179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.732584Z","title":"Masked-attention mask transformer for universal image segmentation","venue":null,"work_id":"8a04b96b-b6b0-4762-915d-1c8e435ed6e6","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.992618Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:a433921d5392ba2dd20161dc49ddcfafd892409bcbf282b695a25c8bb546cb27","observation_id":"dfd3e05b-4488-4f30-9402-a96375d05028","resolution":{"observed_at":"2026-08-12T15:53:34.742847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.691891Z","title":"Reproducible scaling laws for contrastive language-image learning","venue":null,"work_id":"0e836120-e267-4ddd-a73d-f2205b759c13","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.005895Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:7ea852079ade669e6805151a4da3b6b99ee771a4abd18db9d27b5ae20c089088","observation_id":"bfa636b9-7eb7-4292-983e-64c112e2cc83","resolution":{"observed_at":"2026-08-12T15:53:34.704693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.11797","last_updated":"2024-03-31T11:53:55Z","snapshot_observed_at":"2026-08-16T15:46:34.789160Z","submitted_at":"2023-03-21T12:28:21Z","title":"CAT-Seg: Cost Aggregation for Open-Vocabulary Semantic Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.11797","snapshot_observed_at":"2026-08-12T15:53:30.079121Z","title":"Cat-seg: Cost aggregation for open-vocabulary semantic segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.079121Z"},"links":{"cited_paper":"/paper/2303.11797","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:1a3e0d417829de25f2647e2b5c0590d406d2263668f0d564c3965aae406aa4d2","observation_id":"5272cc8f-b55b-45cd-92b1-ba3b62e3fe37","resolution":{"observed_at":"2026-08-12T15:53:30.079121Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2208.08984","last_updated":"2023-06-08T06:35:33Z","snapshot_observed_at":"2026-08-16T16:38:20.473075Z","submitted_at":"2022-08-18T17:55:37Z","title":"Open-Vocabulary Universal Image Segmentation with MaskCLIP","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2208.08984","snapshot_observed_at":"2026-08-12T15:53:30.129614Z","title":"Open- vocabulary panoptic segmentation with maskclip","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.129614Z"},"links":{"cited_paper":"/paper/2208.08984","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:745b9b58b035cba9199470280b8443451357ee7abf5af7b01609a0310912e1ba","observation_id":"5c3e7584-fa90-43a4-8185-0a4fe387b6ed","resolution":{"observed_at":"2026-08-12T15:53:30.129614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-16T09:25:53.087782Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-12T15:53:30.187878Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.187878Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:ea130b23e72a114b1071001d559df765302466451a80bc10c21dbd262e1a99dd","observation_id":"d277024b-19b1-4fc2-b1f8-077edea43f61","resolution":{"observed_at":"2026-08-12T15:53:30.187878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.660755Z","title":"The pascal visual object classes (voc) challenge","venue":null,"work_id":"fbdeedbf-2d56-4b12-9a24-fc6df496e3b1","year":2010},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.241754Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:43b50c1c7b5907069d78d69e0463f10013da7aacf53006dabff25fbf5b5f6fb2","observation_id":"78f240fe-ec8b-4796-aadf-3b4b65160abe","resolution":{"observed_at":"2026-08-12T15:53:34.669246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.619532Z","title":"Scaling up visual and vision-language representa- tion learning with noisy text supervision","venue":null,"work_id":"00de7dda-f664-43ea-a21f-608878938c1e","year":2021},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.253622Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:c1879326d57f2bd4409997ee03484e15533b8f7c72a8e51990ab252ddd71411b","observation_id":"0881ea7d-4ff2-4390-8134-9d8081342a30","resolution":{"observed_at":"2026-08-12T15:53:34.636723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.589301Z","title":"Diffusion models for zero-shot open-vocabulary segmentation","venue":null,"work_id":"c1381a04-b7ae-4b95-9d64-1e12a3778632","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.262474Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:e6a25a581e6ce1e5edaaf5b852147cc765399663135e91433964b51449f38a70","observation_id":"2274c32d-a5ad-4847-8272-79a291cfca52","resolution":{"observed_at":"2026-08-12T15:53:34.594676Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.552160Z","title":"Segment anything","venue":null,"work_id":"1b2f621b-2b91-48d1-8a17-f94a01db2b85","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.323897Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:9c3493007548b9f406d666175c1655164dbac536021fcb274ac308a45aaf0384","observation_id":"6b827282-485e-49ef-88eb-515e84b63026","resolution":{"observed_at":"2026-08-12T15:53:34.561979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.511683Z","title":"ProxyCLIP: Proxy attention improves clip for open-vocabulary segmentation","venue":null,"work_id":"23354389-5a0e-4ec9-a217-32ba836ef273","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.434077Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:57d7990fa6f35bc84c4ac8b75d6dde8f2419c9ffd11b07740dfa0a9198e08ad4","observation_id":"423ec143-ef68-46a0-8dd1-e6d917960360","resolution":{"observed_at":"2026-08-12T15:53:34.527327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.393175Z","title":"ClearCLIP: Decom- posing clip representations for dense vision-language infer- ence","venue":null,"work_id":"81a32e6a-df33-4468-83de-8fa7f72f0d47","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.511695Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:9eef5a57e51a5aaec9cc3e7f208c18c135dd960ae995775de352dd299a858749","observation_id":"6dc9b584-b0b6-4919-87f8-cb622f46ef15","resolution":{"observed_at":"2026-08-12T15:53:34.469805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.340745Z","title":"Anti- adversarially manipulated attributions for weakly and semi- supervised semantic segmentation","venue":null,"work_id":"7e355020-f664-4cb0-99d8-d1144f30761b","year":2021},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.524335Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:aeb464e16e297e853e2ff63a93a8750dcad0c054ac8b7f1a9c8d3b8bdcd3ae17","observation_id":"0e733d73-93f9-4fc4-b743-8420134f2b27","resolution":{"observed_at":"2026-08-12T15:53:34.352659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05653","last_updated":"2024-09-16T09:10:00Z","snapshot_observed_at":"2026-08-16T15:41:05.975016Z","submitted_at":"2023-04-12T07:16:55Z","title":"A Closer Look at the Explainability of Contrastive Language-Image Pre-training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05653","snapshot_observed_at":"2026-08-12T15:53:30.578203Z","title":"A closer look at the explainability of contrastive language-image pre-training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.578203Z"},"links":{"cited_paper":"/paper/2304.05653","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:3c0a4162ba7ce6fc7685412f730619e42fc5912dda00c66327493125556781fa","observation_id":"9fa5cd2c-22c7-4835-9522-776468f09def","resolution":{"observed_at":"2026-08-12T15:53:30.578203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.294257Z","title":"Open-vocabulary semantic segmentation with mask-adapted clip","venue":null,"work_id":"0258942c-3945-4216-93e8-ffba48fd9683","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.672403Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:20325ba161aa8b7f588e82f5875dad795d985ad1822f04ad65a0327a53e6bff3","observation_id":"29a748cd-ff89-4e14-8ada-6c5fa9bea170","resolution":{"observed_at":"2026-08-12T15:53:34.308320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.262440Z","title":"CLIP is also an efficient segmenter: A text-driven approach for weakly su- pervised semantic segmentation","venue":null,"work_id":"45c99e5d-d3cc-43e0-be57-938ea56bbd3c","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.744405Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:d9e4b37151195e5b1d6914e41abc27e2eea87301d043a191e82099b0d4110228","observation_id":"cafa38ba-9d28-4b8a-a449-f375d9a8c235","resolution":{"observed_at":"2026-08-12T15:53:34.274869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.227983Z","title":"TagCLIP: A local-to-global framework to enhance open-vocabulary multi-label classification of clip without training","venue":null,"work_id":"cdfde013-3cf6-4ac1-b38c-c6d32e9c8ef5","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.756293Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:38aa4e74a906cb1f35aaac5393a92fdffd916ffe79df401814bde9432cc9f76f","observation_id":"449e98ac-1963-48a5-babc-1bd3c2baeffb","resolution":{"observed_at":"2026-08-12T15:53:34.237480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.194916Z","title":"Fully convolutional networks for semantic segmentation","venue":null,"work_id":"3829924a-c49c-410f-ab57-66614bdfdbd6","year":2015},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.764526Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:fce4eb0f772050975efa4528a6be5dff7f9287a7bee0f0bc6ea8b0f713477c1c","observation_id":"9744087a-2167-4d00-b9a8-cbc741bbe050","resolution":{"observed_at":"2026-08-12T15:53:34.201120Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.157370Z","title":"SegCLIP: Patch aggregation with learnable centers for open-vocabulary semantic segmentation","venue":null,"work_id":"09021794-b224-4362-bd5a-8f188af55c9b","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.854185Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:d4cfbab16145915ebe6787768e97b5be0e3a73498cfda21536b44ca4b40ff0a3","observation_id":"66449b3c-c91c-4a61-8ca8-18a167b6a519","resolution":{"observed_at":"2026-08-12T15:53:34.169300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.113034Z","title":"The role of context for object detection and semantic segmentation in the wild","venue":null,"work_id":"81b8cb23-2e23-43b3-b689-e85ced8155bb","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.956732Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:423f445e0ccac8c3347fe01bd03274882f17fb67addc129ea68f4181e86df22b","observation_id":"4cbf94d9-3b2a-4eb4-9e34-5ee0d946aa24","resolution":{"observed_at":"2026-08-12T15:53:34.127614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.901696Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"1e5bf99f-3473-4987-a9a6-fb7ffb2c0d49","year":2021},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.980568Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:79ab7e34c0d35108e26ff380ad8cf3f3ec5320310cdc909e4fd2f5cc74a08211","observation_id":"82a288b3-01c7-41c6-9ee1-579ecf5ed70b","resolution":{"observed_at":"2026-08-12T15:53:33.983972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.851467Z","title":"Per- ceptual grouping in contrastive vision-language models","venue":null,"work_id":"2f8f1d38-6578-43b9-8eb4-9341f8d99f27","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.004171Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:4c1b4e10ea309fa991e3c77a1506cbfa83a6dcd94be68d3aa1cffd2248ff42e6","observation_id":"d3175b13-b7f6-4075-a543-9417b6f3b717","resolution":{"observed_at":"2026-08-12T15:53:33.861006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.807847Z","title":"ViewCo: Discovering text-supervised segmentation masks via multi-view semantic consistency","venue":null,"work_id":"c6cd168e-6473-4eae-a4dc-d78e3281375a","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.016261Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:9869f262b43afd4dfeba71f81d5df21ca1b3a2da63d5698d5628e6d05218d319","observation_id":"e7ea2c76-8fe9-4204-97ac-b02c22479857","resolution":{"observed_at":"2026-08-12T15:53:33.818391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.773559Z","title":"High-resolution image syn- thesis with latent diffusion models","venue":null,"work_id":"ee6deaff-52c1-49b5-81e4-df93aff0fc0c","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.091640Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:13b0648c86e40dd06687ce2ca61e2eab9a1d53eb081e4f14b959800caabe6759","observation_id":"dc4bb987-e0bd-48da-8858-9c13f054bca6","resolution":{"observed_at":"2026-08-12T15:53:33.785884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.738026Z","title":"To- ken contrast for weakly-supervised semantic segmentation","venue":null,"work_id":"df8e0a80-ca45-4558-ba35-fe12adc86641","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.205206Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:4903397eb2f8386bd029061472c5adeb7ac1d2d6cb528bdb456ee24bf6d28fef","observation_id":"8db690b3-cad4-42db-9946-f1bf9512de3a","resolution":{"observed_at":"2026-08-12T15:53:33.749028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.702716Z","title":"Laion-5B: An open large-scale dataset for training next generation image-text models","venue":null,"work_id":"faf43ba7-d316-44af-8190-1740da79bb5d","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.270985Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:4a5b84420098f805e161cb9e51300595bd56461974ba76e9620d77072c341b11","observation_id":"ff6fa4f0-1142-43e1-ac19-a7faab042712","resolution":{"observed_at":"2026-08-12T15:53:33.713015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.662385Z","title":"ReCo: Re- trieve and co-segment for zero-shot transfer","venue":null,"work_id":"b4c5bc70-8983-455a-80a4-45323cf79ada","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.288916Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:6105e7bf93d587c39234af061ac10f9132b81f87fd5bdafd04082db32b7263b6","observation_id":"3ef15bd6-675a-46e3-ae5b-d66e8b41b6db","resolution":{"observed_at":"2026-08-12T15:53:33.672270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.03209","last_updated":"2024-10-08T08:32:36Z","snapshot_observed_at":"2026-08-16T22:46:12.892385Z","submitted_at":"2024-09-05T03:07:26Z","title":"iSeg: An Iterative Refinement-based Framework for Training-free Segmentation","version":4},"cited_work":{"arxiv_id":"2409.03209","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.03209","snapshot_observed_at":"2026-08-12T15:53:32.117301Z","title":"iSeg: An Iterative Refinement-based Framework for Training-free Segmentation","venue":"cs.CV","work_id":"7ce38b09-3ecc-408e-a74e-e078db4a6b0c","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.299753Z"},"links":{"cited_paper":"/paper/2409.03209","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:5df110a79dbd293b6a48e4c253bed5ce8198f84acab3f9bf13131158411e4eb6","observation_id":"76af3956-dcb7-43e3-82b4-dade04414c75","resolution":{"observed_at":"2026-08-12T15:53:32.128129Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.624187Z","title":"CLIP as RNN: Segment countless visual concepts with- out training endeavor","venue":null,"work_id":"a1026133-8569-40cb-9ab1-c2e9613b3bc1","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.324636Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:ec5581a1b535f017dbe69def1bb074a272fb25f5c85510e65b53a1848718d372","observation_id":"ee87535e-4d4b-41ad-8387-f8c20bdb7c73","resolution":{"observed_at":"2026-08-12T15:53:33.634456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.462816Z","title":"Sclip: Rethinking self-attention for dense vision-language inference","venue":null,"work_id":"cbe222b4-971f-4a1b-a813-68f41181ef87","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.421744Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:bbd7681bc08449c106a4d33c76b85eb02f212bfc5bb71d56c3bffe4a4e926b04","observation_id":"0e62b1ef-f7ee-410a-9bee-51decc852a1f","resolution":{"observed_at":"2026-08-12T15:53:33.574412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.386515Z","title":"Sam-clip: Merging vision foundation models towards semantic and spatial understanding","venue":null,"work_id":"3df78d9e-8e00-4d0f-aa5d-3a3804f651c3","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.466108Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:d5c14f220406b87f4d221f5e37fe98cf891071a2db83f687748372f6ced0b240","observation_id":"537c9fe7-9180-47fc-beb5-44f643700f30","resolution":{"observed_at":"2026-08-12T15:53:33.398863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.02773","last_updated":"2024-01-22T07:18:55Z","snapshot_observed_at":"2026-08-16T15:02:56.591450Z","submitted_at":"2023-09-06T06:31:08Z","title":"Diffusion Model is Secretly a Training-free Open Vocabulary Semantic Segmenter","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.02773","snapshot_observed_at":"2026-08-12T15:53:31.546345Z","title":"Diffusion model is secretly a training-free open vocabulary semantic segmenter","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.546345Z"},"links":{"cited_paper":"/paper/2309.02773","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:30f771aca6ca0edfec81425ec6f246da72c56dcdd4ff5c67296708c9b6ac9a12","observation_id":"f1e187ce-2dfd-4150-8125-3768a669df13","resolution":{"observed_at":"2026-08-12T15:53:31.546345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.232759Z","title":"Clip-dinoiser: Teaching clip a few dino tricks for open- vocabulary semantic segmentation","venue":null,"work_id":"3c002ebf-4a5a-4187-a87c-495ebc334456","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.678102Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:d67fc28ee489cabf731107a5b425a6d8c74244d91285ceb3c21a290d168e75a7","observation_id":"e021f2c3-832d-4647-886f-e9733f0fac57","resolution":{"observed_at":"2026-08-12T15:53:33.361545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.04109","last_updated":"2024-10-01T10:30:07Z","snapshot_observed_at":"2026-08-16T15:02:21.597744Z","submitted_at":"2023-09-08T04:10:01Z","title":"From Text to Mask: Localizing Entities Using the Attention of Text-to-Image Diffusion Models","version":2},"cited_work":{"arxiv_id":"2309.04109","doi":null,"metadata_source":"pith","pith_arxiv_id":"2309.04109","snapshot_observed_at":"2026-08-12T15:53:32.028523Z","title":"From Text to Mask: Localizing Entities Using the Attention of Text-to-Image Diffusion Models","venue":"cs.CV","work_id":"6db47440-5bd3-4a74-a2af-14863c4d90c0","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.718299Z"},"links":{"cited_paper":"/paper/2309.04109","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:793c3ed99f9e2626b4c46ef93ef80cdeaa3a4003394726f66fbfd71a281b4d27","observation_id":"5268430a-a04e-4686-bde8-0acc42f79d99","resolution":{"observed_at":"2026-08-12T15:53:32.041705Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.194200Z","title":"Sed: A simple encoder-decoder for open- vocabulary semantic segmentation","venue":null,"work_id":"8d4f1c10-f1db-4a1d-9b0a-c545b5abf82a","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.738935Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:669a1eff873773062550ffc59d8bb3a68ec008eff88a2d4bf2b2dd1318a3023b","observation_id":"0fd812d0-79ab-49e5-8126-9f64443fd613","resolution":{"observed_at":"2026-08-12T15:53:33.203098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.088063Z","title":"Alvarez, and Ping Luo","venue":null,"work_id":"c9abf15e-04d4-41e8-9710-31acae16f91a","year":2021},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.752197Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:ab1d0e98140f05c88b937e2b47493ea5574d5139a77045d014e93c7e57067f53","observation_id":"6818b0d7-7fbb-4083-b73b-4807879449f2","resolution":{"observed_at":"2026-08-12T15:53:33.169597Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.940660Z","title":"Clims: Cross language image matching for weakly supervised se- mantic segmentation","venue":null,"work_id":"aee72a42-05d7-4945-b6f3-bf2bc1c4ebc5","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.765289Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:0da06631e11bb176d836a6ebe27c77da819a9f58c4bc085e3a30e0d7a67847fb","observation_id":"82e486ea-48b1-475c-bb02-2c9994ba1305","resolution":{"observed_at":"2026-08-12T15:53:32.990468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.892729Z","title":"Rewrite caption semantics: Bridging seman- tic gaps for language-supervised semantic segmentation","venue":null,"work_id":"c103edb7-5e7a-44fe-95ce-474df54aeef8","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.782187Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:41448ff405b30eca7176264403a7540d9f680d97c1dcde67e7c71a626b3bc8e7","observation_id":"179b678d-4304-41da-b23a-b38123aa2586","resolution":{"observed_at":"2026-08-12T15:53:32.915948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.841835Z","title":"Groupvit: Semantic segmentation emerges from text supervision","venue":null,"work_id":"26f961a9-ab64-4f4b-b686-8f7d6952bd31","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.798853Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:046d8661b9ccd0b91915455f0de5de07470d07012de0123f764a7c717b52cad7","observation_id":"aa91ddfa-5545-4106-83cb-1dbe4aaf9647","resolution":{"observed_at":"2026-08-12T15:53:32.850323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.800365Z","title":"Learning open-vocabulary seman- tic segmentation models from natural language supervision","venue":null,"work_id":"75c7f5ed-9018-44c5-a027-12ef66c5513c","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.808098Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:2d971f7afa32af6e0af809051bbee7e99d8ee99f93e39b4a789a317c0a2b2c43","observation_id":"34810fe9-a51d-4920-83db-c970a2641733","resolution":{"observed_at":"2026-08-12T15:53:32.809151Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.763433Z","title":"Open-vocabulary panoptic segmentation with text-to-image diffusion models","venue":null,"work_id":"f5a63cd0-ac4c-4341-be83-f2f8644fc4d7","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.817712Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:2b49c0fcac78d5ae386bd016749a22c6b678993b5c4f7d420089e29c93fdefab","observation_id":"2f00c189-f6a3-4241-a794-bda1d061773d","resolution":{"observed_at":"2026-08-12T15:53:32.773430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.693071Z","title":"Multi-class token transformer for weakly supervised se- mantic segmentation","venue":null,"work_id":"2e2e7efe-df80-4b01-94ba-d4b13b09352c","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.824494Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:ad66e20cde9f61a5038abb0efe49c00293769c78518050e3f35f7fed0e67d241","observation_id":"16223171-313d-4031-a611-a09d431217d6","resolution":{"observed_at":"2026-08-12T15:53:32.726561Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.647886Z","title":"Side adapter network for open-vocabulary semantic segmentation","venue":null,"work_id":"69885f3f-bfcc-414e-aac8-63b47a6c6626","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.838226Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:b79d0b35b664527229d58656d06165bf913cfddcf729c9631a14edfcb6d5fd12","observation_id":"0732b187-27e6-4bd9-9ecd-79a5528df144","resolution":{"observed_at":"2026-08-12T15:53:32.660574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.02487","last_updated":"2023-11-14T19:10:49Z","snapshot_observed_at":"2026-08-16T15:10:43.081176Z","submitted_at":"2023-08-04T17:59:01Z","title":"Convolutions Die Hard: Open-Vocabulary Segmentation with Single Frozen Convolutional CLIP","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.02487","snapshot_observed_at":"2026-08-12T15:53:31.853023Z","title":"Convolutions die hard: Open-vocabulary seg- mentation with single frozen convolutional clip","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.853023Z"},"links":{"cited_paper":"/paper/2308.02487","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:83cf666cdff0cdb6aede4470af5bcf7f16f6a7db6f1d7364777094573f759003","observation_id":"dedbb98a-65d2-424e-b811-32470ae6b714","resolution":{"observed_at":"2026-08-12T15:53:31.853023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.588332Z","title":"Open vocabulary scene parsing","venue":null,"work_id":"f8a3eb35-17d5-4c6c-a233-193beff8b4f2","year":2002},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.865201Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:6c81b2c9b8a702d89e2e294b0eae29eec953197f4a8d3146dd13a5dc437f7e1c","observation_id":"91ee6cf4-fc84-47f8-bc5f-272acd54846b","resolution":{"observed_at":"2026-08-12T15:53:32.616043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:31.878658Z","title":"Semantic under- standing of scenes through the ade20k dataset","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.878658Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:505f7975869a41132522a972eabd6f96e978911d1f6838aa1e2d930ba649383d","observation_id":"c5655f20-8b1e-4db0-8adb-12423fbf6252","resolution":{"observed_at":"2026-08-12T15:53:31.878658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.401277Z","title":"Extract free dense labels from clip","venue":null,"work_id":"dc71af18-c840-4af3-bc7e-87385c42e1b3","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.889301Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:2e653beda7125a1639a7b6fc4f1135bca3f30a88a72e68bd3941cee3ce750956","observation_id":"3241dc5e-106a-4fec-82df-4def85422460","resolution":{"observed_at":"2026-08-12T15:53:32.461065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T16:27:47.281160Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":2,"verified_fuzzy":42},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 2 inbound Pith citation observations for arXiv:2411.13836."}