{"as_of":"2026-08-16T17:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a8ba19db095e39457aa9715230de287f095af11ac31d2792bcd883b0b5c5d3e0","coverage":[{"denominator":66,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T23:50:23.565560Z","state":"measured"},{"denominator":66,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":66,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.16058/citation-record","integrity":"/paper/2506.16058/integrity","json":"/paper/2506.16058/citation-record.json","paper":"/paper/2506.16058"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:23.250219Z","title":"Self-calibrated clip for training-free open-vocabulary segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.250219Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:e6c361bbcee290fc565c5219a95eeccbdf990db1d8a4988202196e9629190fe8","observation_id":"a97559ff-9004-4296-ab05-ec06c700efd3","resolution":{"observed_at":"2026-08-06T23:50:23.250219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14231","last_updated":"2025-05-20T11:40:43Z","snapshot_observed_at":"2026-08-12T14:22:37.782293Z","submitted_at":"2025-05-20T11:40:43Z","title":"UniVG-R1: Reasoning Guided Universal Visual Grounding with Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14231","snapshot_observed_at":"2026-08-06T23:50:23.255593Z","title":"Univg-r1: Reasoning guided universal visual grounding with reinforce- ment learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.255593Z"},"links":{"cited_paper":"/paper/2505.14231","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:06ce8f755492d89b1a7a2819adf347152204ba8edf5d4b72aa81d8ba39fbe2ca","observation_id":"8ddce7f4-5588-41a2-83f5-52828cd4f189","resolution":{"observed_at":"2026-08-06T23:50:23.255593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.721296Z","title":"Brostow, Jamie Shotton, Julien Fauqueur, and Roberto Cipolla","venue":null,"work_id":"3c602268-fabe-401c-a0b7-c484e22fb9ba","year":2008},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.261190Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:19e188b76962c9a818e4e6e1a9dbc1db5dd49b777877e569d8a66cd776fc35a7","observation_id":"97e86502-505d-4e63-a351-faf64d93292e","resolution":{"observed_at":"2026-08-06T23:50:24.726791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.706160Z","title":"Zero-shot semantic segmentation","venue":null,"work_id":"d4ecdf32-e5a4-4e69-9db8-9624b33576b4","year":2019},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.266818Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:03d91f53c7edac4dc84e8f9c05cd022fac57de5081c6d822bf8cc28d8eac9831","observation_id":"668c0a39-61f2-4d63-b2c7-837a4f063b19","resolution":{"observed_at":"2026-08-06T23:50:24.711443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.691013Z","title":"End-to- end object detection with transformers","venue":null,"work_id":"eae58f91-8ea4-4176-809e-b0b456964c6e","year":2020},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.271584Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:a18705df6c84e30f3f3ddba80931cbfd79850572f424a579df9d7a40d5fef7f2","observation_id":"c50bfe99-7026-4e63-8648-f42be4d18a4a","resolution":{"observed_at":"2026-08-06T23:50:24.696514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.676901Z","title":null,"venue":null,"work_id":"51a0d026-7010-42bb-8cc6-80e11693b1af","year":2018},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.276968Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:d9b42cc901035dc1a8edbeb2b6e005f83643e7bf1d803ccd98e605252078381e","observation_id":"e38a447f-63b6-4c93-95d4-88c76ddd1afe","resolution":{"observed_at":"2026-08-06T23:50:24.681255Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11740","last_updated":"2020-07-17T22:19:59Z","snapshot_observed_at":"2026-08-09T19:02:25.484253Z","submitted_at":"2019-09-25T20:02:54Z","title":"UNITER: UNiversal Image-TExt Representation Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11740","snapshot_observed_at":"2026-08-06T23:50:23.281943Z","title":"UNITER: learning universal image-text representations","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.281943Z"},"links":{"cited_paper":"/paper/1909.11740","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:125cbcb3f57c2fe8beeeaea4137fec5b2a8d840dcae6545851cd26f54682c2b9","observation_id":"97cb7be6-a06c-463a-8359-e2eb0820f02c","resolution":{"observed_at":"2026-08-06T23:50:23.281943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10764","last_updated":"2021-12-20T18:59:59Z","snapshot_observed_at":"2026-08-13T17:11:52.200632Z","submitted_at":"2021-12-20T18:59:59Z","title":"Mask2Former for Video Instance Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10764","snapshot_observed_at":"2026-08-06T23:50:23.286782Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.286782Z"},"links":{"cited_paper":"/paper/2112.10764","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:76446ca75fcc476690290abc04adf479beae608687dc859af74f23454c08baf3","observation_id":"cab951f1-9f45-4213-978b-f68eea78264e","resolution":{"observed_at":"2026-08-06T23:50:23.286782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:23.292230Z","title":"Schwing, Alexan- der Kirillov, and Rohit Girdhar","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.292230Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:ccca4777078e01b5bd9ec77830773c0356ef3fea3cce60887238a542c545af4f","observation_id":"c5b765f5-6d38-48ca-89e7-f31e3eecfc23","resolution":{"observed_at":"2026-08-06T23:50:23.292230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.651884Z","title":"Cat-seg: Cost aggregation for open-vocabulary semantic segmenta- tion","venue":null,"work_id":"a88d2659-2b35-4136-8147-1b5fed565f48","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.297117Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:f75f46ef58a00153810273c5e258d8fe7771063b02f56151bdc68e470c43f877","observation_id":"d3f43add-dd73-459f-9379-39f466b93474","resolution":{"observed_at":"2026-08-06T23:50:24.656705Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:23.301829Z","title":"The cityscapes dataset for semantic urban scene understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.301829Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:56b9f042e8eb36e6b5d252474219b9e0f9906d4592a8c51a2316f5a09a6760ce","observation_id":"4723c5f7-8b26-4713-a00e-80203d198265","resolution":{"observed_at":"2026-08-06T23:50:23.301829Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.626285Z","title":"De- coupling zero-shot semantic segmentation","venue":null,"work_id":"8dbd26db-ac5a-4fa3-952b-5c8c85274426","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.307089Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:181ee4e140cb4192f00edc0efa683e90e79ec4db8ae2da0b5235f5986af31441","observation_id":"e4678ec7-af9e-4dda-ac7b-b7463e10667d","resolution":{"observed_at":"2026-08-06T23:50:24.630931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.610530Z","title":"The pascal visual object classes challenge: A retrospective.IJCV, 111:98–136, 2015","venue":null,"work_id":"d17ccd2f-0342-481b-8f4b-142b157c099a","year":2015},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.311413Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:9fed06ba168f17251959336f19585575eb429566d48f5d2e53e6a5226e591e8f","observation_id":"9f2bc313-d6c4-40a1-8738-f963b9a62445","resolution":{"observed_at":"2026-08-06T23:50:24.615216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.595394Z","title":"Large-scale unsu- pervised semantic segmentation","venue":null,"work_id":"be58d233-6201-4abc-9c46-b698b1cdc4f0","year":null},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.316014Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:0edbc88d1c7f67307a266e673a69197a924d63f4fef7c80a4bccd4a69bf10c5b","observation_id":"0d1d0ab4-f453-41f8-a227-bfef21cb9f24","resolution":{"observed_at":"2026-08-06T23:50:24.600200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.12143","last_updated":"2022-07-20T21:56:52Z","snapshot_observed_at":"2026-08-13T17:09:52.986186Z","submitted_at":"2021-12-22T18:57:54Z","title":"Scaling Open-Vocabulary Image Segmentation with Image-Level Labels","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.12143","snapshot_observed_at":"2026-08-06T23:50:23.320589Z","title":"Open-vocabulary image segmentation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.320589Z"},"links":{"cited_paper":"/paper/2112.12143","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:b3b39a30ed4dd4549ce0a6f6d8a3bf9e8c5f532d0b8b89c3645d3449ddfda646","observation_id":"edc5a565-09b0-4bc9-8784-dc05196a8f99","resolution":{"observed_at":"2026-08-06T23:50:23.320589Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.578767Z","title":"Random walks for image segmentation","venue":null,"work_id":"a6b56625-211c-43a8-a9cb-601a99e14347","year":2006},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.325420Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:db26ea5d85af71e326825a811c25bdabbfed9469ff9751d380525b1dba126bdb","observation_id":"9e39c27a-a7ff-4470-9077-2fd5cb6a8031","resolution":{"observed_at":"2026-08-06T23:50:24.584747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.562936Z","title":"Global knowledge calibration for fast open-vocabulary segmentation","venue":null,"work_id":"d432bfff-b32d-424c-b80e-d07d176da4d3","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.330380Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:032eaf953d868bdbeb9815050b26206ad6cdc7f8ce1049cea5caf6d13231a10f","observation_id":"39bb6abf-29bb-4c4f-8712-ea9fb4931231","resolution":{"observed_at":"2026-08-06T23:50:24.567916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.547262Z","title":"Primitive gener- ation and semantic-related alignment for universal zero-shot segmentation","venue":null,"work_id":"76c89121-5761-453e-8151-85147174b4a0","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.335212Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:f0d5f3e1999e878d8493e76d02ce33d55de3feab7ab95bc65b528ee1bd322503","observation_id":"25db6e78-78b1-40f8-b904-20340d80d6c3","resolution":{"observed_at":"2026-08-06T23:50:24.552376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:23.340138Z","title":"Planning-oriented autonomous driving","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.340138Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:fa9571ce3dbc737a56fde6a6e3a67335ce2824fc149385dc702833632759b85d","observation_id":"b676ed18-8618-4d5d-a61e-2ef7499ba897","resolution":{"observed_at":"2026-08-06T23:50:23.340138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08580","last_updated":"2025-01-15T05:00:03Z","snapshot_observed_at":"2026-08-15T17:40:44.350522Z","submitted_at":"2025-01-15T05:00:03Z","title":"Densely Connected Parameter-Efficient Tuning for Referring Image Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08580","snapshot_observed_at":"2026-08-06T23:50:23.344365Z","title":"Densely connected parameter- efficient tuning for referring image segmentation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.344365Z"},"links":{"cited_paper":"/paper/2501.08580","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:d1bfde665f738127e784272f2a64f74c3b31b4f91ea289b6c77444732c3a8ac3","observation_id":"3268e0aa-7e79-441c-93b2-a1f7b2c8fd6c","resolution":{"observed_at":"2026-08-06T23:50:23.344365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00869","last_updated":"2024-03-26T12:56:55Z","snapshot_observed_at":"2026-08-16T14:38:22.570490Z","submitted_at":"2023-12-01T19:00:17Z","title":"Segment and Caption Anything","version":2},"cited_work":{"arxiv_id":"2312.00869","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.00869","snapshot_observed_at":"2026-08-06T23:50:23.790350Z","title":"Segment and Caption Anything","venue":"cs.CV","work_id":"2e7cc23b-8dc0-4956-9b7e-1b93f125232d","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.348944Z"},"links":{"cited_paper":"/paper/2312.00869","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:8798572dfea393291dc0258053c94ab504a9cf43f40adbbbd84dee84d6cf181f","observation_id":"de7f347b-2b56-4e45-98e4-5c4184e9f8af","resolution":{"observed_at":"2026-08-06T23:50:23.795324Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.520741Z","title":"Proxydet: Synthesizing proxy novel classes via classwise mixup for open-vocabulary object detection","venue":null,"work_id":"dd89ade7-4a95-4900-a548-0a3f1d06e255","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.354116Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:510481dee1ae5e66574a4adff16f51e024ed75d8718174d473d483aec9118dcb","observation_id":"cc55fbdc-facc-4b62-86d5-2b7d9f022012","resolution":{"observed_at":"2026-08-06T23:50:24.525350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.506797Z","title":"Le, Yun-Hsuan Sung, Zhen Li, and Tom Duerig","venue":null,"work_id":"d9bafb52-259d-40ab-80df-54e27cf7838b","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.359615Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:f274e23259a0d81547f86c224c91e63d41449186e8d7d52e3d5b7b396ce6aab1","observation_id":"98fa837c-4a11-4694-8f06-5a56b65b9df6","resolution":{"observed_at":"2026-08-06T23:50:24.511326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.490349Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision","venue":null,"work_id":"9b8e06e5-98ca-41ff-92f8-174d131aa7aa","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.364378Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:b1c1db394002133258e4595fa311e965656bc79dcb79f8db8d9ce8d66e87da22","observation_id":"d147be5f-999c-4b9c-a8d1-e16f67e5a740","resolution":{"observed_at":"2026-08-06T23:50:24.495762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00240","last_updated":"2023-09-30T03:27:31Z","snapshot_observed_at":"2026-08-16T14:56:19.896324Z","submitted_at":"2023-09-30T03:27:31Z","title":"Learning Mask-aware CLIP Representations for Zero-Shot Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00240","snapshot_observed_at":"2026-08-06T23:50:23.369382Z","title":"Learning mask-aware clip repre- sentations for zero-shot segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.369382Z"},"links":{"cited_paper":"/paper/2310.00240","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:8013bf004935c6f1fcf9c2e1a6f3d3c6fe3c085bb1b450e1888716b7a72d90de","observation_id":"8e05eb0c-bd88-40b9-a4bf-6d017e7c101b","resolution":{"observed_at":"2026-08-06T23:50:23.369382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.475008Z","title":"Collaborative vision-text rep- resentation optimizing for open-vocabulary segmentation","venue":null,"work_id":"4fa75ffb-a5a5-4ae4-beca-77423c9d00cc","year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.374458Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:5e856295a268db4976fa95f9770d746c7f2d4e8f323ffaea710aeb0850694df3","observation_id":"1fc46008-179d-4f55-97b9-d16c91fbdb51","resolution":{"observed_at":"2026-08-06T23:50:24.479936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.459136Z","title":"Weinberger, Serge J","venue":null,"work_id":"35f2523b-ccfc-4486-b656-d412dd3d921a","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.379399Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:394577d02f043bd112d8288c72378bc9dd5cfbb725b77d7394c625c0212aacbc","observation_id":"498432c4-95c4-45e0-855e-1e3a1fe63fd2","resolution":{"observed_at":"2026-08-06T23:50:24.464721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.443324Z","title":"Unicoder-vl: A universal encoder for vision and lan- guage by cross-modal pre-training","venue":null,"work_id":"222f19c9-898d-44eb-bd59-ad8e348981be","year":2020},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.384236Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:b3f591da53ef8690b8a158d837900bcc24f771a02fa9cd118defa896447380da","observation_id":"0a889458-a699-4ddb-8b2b-e8528c40033f","resolution":{"observed_at":"2026-08-06T23:50:24.448213Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.427053Z","title":"Ordinalclip: Learning rank prompts for language-guided ordinal regression","venue":null,"work_id":"fb6117ec-986f-4985-bb3d-4b5d302a877a","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.388956Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:b245b6e76f19ef46ae93afe7cfe712d3b2937d736bcf81638a9a3b3ea3528b02","observation_id":"0ebd4417-b1aa-4019-927d-55a82dc79e9f","resolution":{"observed_at":"2026-08-06T23:50:24.432092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.411347Z","title":"Oscar: Object-semantics aligned pre-training for vision-language tasks","venue":null,"work_id":"2844ab83-fe87-4e64-af11-bf642b407a84","year":null},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.393647Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:33f4b4fcd33d9e4bb3070a48bf34b327baba3857c264df5afa4dd233381875c2","observation_id":"c811938a-aac7-4ac5-8e09-acf2ef80452c","resolution":{"observed_at":"2026-08-06T23:50:24.416066Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.04150","last_updated":"2023-04-01T19:00:47Z","snapshot_observed_at":"2026-08-16T16:25:50.641792Z","submitted_at":"2022-10-09T02:57:32Z","title":"Open-Vocabulary Semantic Segmentation with Mask-adapted CLIP","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.04150","snapshot_observed_at":"2026-08-06T23:50:23.398271Z","title":"Open-vocabulary semantic segmentation with mask-adapted CLIP","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.398271Z"},"links":{"cited_paper":"/paper/2210.04150","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:6fccf05a1495c7d6ae1dc57262ee5ea5316f825223690805c708ede0c90cbd39","observation_id":"68a2fe49-d63e-45c2-bf79-9736b651e201","resolution":{"observed_at":"2026-08-06T23:50:23.398271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.395676Z","title":"Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll ´ar, and C","venue":null,"work_id":"abede16f-a3b0-4153-9517-b74bdefba68b","year":2014},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.402939Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:ded586e89588caeb7c05f5c7ae7a3cbad599302613eade2b43ebc12c80524619","observation_id":"fd7982dc-77c4-4d1e-ab7a-7ab4b057b9c1","resolution":{"observed_at":"2026-08-06T23:50:24.401125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.379979Z","title":"Quality- aware and selective prior enhancement memory network for video object segmentation","venue":null,"work_id":"9e684047-ed41-4df1-837f-af37a5cf840e","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.407360Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:61a243f7c548ca0580bfd9ffabcc7c8f5c3972be17cd7923a33cec38d7d38252","observation_id":"6efe546e-7a4c-4061-8f09-b8ab3b5e83b7","resolution":{"observed_at":"2026-08-06T23:50:24.385247Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.365441Z","title":"Global spectral filter memory network for video object segmentation","venue":null,"work_id":"2a8a18d8-199f-4e97-aaf6-d1dcb8cc2890","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.411715Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:9a949a3e89c97f819cdc0fd8b673ff357906b56fe102d5ee7e567006369ad7e3","observation_id":"66b8b1e0-6604-4c24-998d-108842b10359","resolution":{"observed_at":"2026-08-06T23:50:24.370046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.349863Z","title":"Learning quality-aware dynamic memory for video object segmentation","venue":null,"work_id":"351e3ea7-12b3-4270-b3ca-84a219e1a1e1","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.416391Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:f4e9f2094963cd400221b1f23dee5aef655d1880b37c905895c6b367875ed469","observation_id":"05c253fb-20d2-474e-b290-67a9a5a52fb9","resolution":{"observed_at":"2026-08-06T23:50:24.355059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.01623","last_updated":"2024-11-26T14:03:51Z","snapshot_observed_at":"2026-08-16T14:38:00.048273Z","submitted_at":"2023-12-04T04:47:48Z","title":"Universal Segmentation at Arbitrary Granularity with Language Instruction","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.01623","snapshot_observed_at":"2026-08-06T23:50:23.420753Z","title":"Universal segmentation at arbi- trary granularity with language instruction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.420753Z"},"links":{"cited_paper":"/paper/2312.01623","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:1e7f908a5b4d21199ae622bc11c77e6303acdfbd2f2654aad8ebc47115cbafea","observation_id":"4568aa7d-2900-49b7-a87c-311f3614fb37","resolution":{"observed_at":"2026-08-06T23:50:23.420753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.334546Z","title":"Open-vocabulary segmentation with semantic-assisted calibration","venue":null,"work_id":"e1983188-589b-4ed4-9986-c40d99568515","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.426153Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:32ecdc25d00e9630800a255c777fd54f643e6bcbe067a4e5895440c052914147","observation_id":"baedb819-8184-4044-9af5-e317f128b7de","resolution":{"observed_at":"2026-08-06T23:50:24.340256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.320027Z","title":"Learning high-quality dynamic memory for video object segmentation","venue":null,"work_id":"ecd9e43d-39b0-446c-af13-0d6d73b2c4b2","year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.430987Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:407b0abb6ab1c5a67bcaeae5fdf38cde186fb4545536b523575df05d0ccbe83d","observation_id":"0b47705e-28cb-468b-a303-fe33b669b1b1","resolution":{"observed_at":"2026-08-06T23:50:24.324665Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.07062","last_updated":"2023-12-14T03:28:49Z","snapshot_observed_at":"2026-08-16T14:35:36.159530Z","submitted_at":"2023-12-12T08:30:09Z","title":"ThinkBot: Embodied Instruction Following with Thought Chain Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.07062","snapshot_observed_at":"2026-08-06T23:50:23.435750Z","title":"Thinkbot: Embodied instruction fol- lowing with thought chain reasoning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.435750Z"},"links":{"cited_paper":"/paper/2312.07062","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:c51eb27c0ba10f0748b0428cd7d4b852e07c70ada0291519a009a002293266c9","observation_id":"4358f59c-72f4-4cdf-905f-4afcf8fea044","resolution":{"observed_at":"2026-08-06T23:50:23.435750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.304739Z","title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","venue":null,"work_id":"04cd0639-a499-47fc-afe0-570cd6f6f343","year":2019},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.440778Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:72296d7f7a34a6b7a7051d78bd95ae100c72036fdbe21ad42b7dbf43b546e015","observation_id":"1b876d93-17fc-45af-9517-fa78d3341338","resolution":{"observed_at":"2026-08-06T23:50:24.310125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17011","last_updated":"2023-05-26T15:13:44Z","snapshot_observed_at":"2026-08-16T15:29:16.469299Z","submitted_at":"2023-05-26T15:13:44Z","title":"SOC: Semantic-Assisted Object Cluster for Referring Video Object Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.17011","snapshot_observed_at":"2026-08-06T23:50:23.445973Z","title":"Soc: Semantic-assisted object cluster for referring video object segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.445973Z"},"links":{"cited_paper":"/paper/2305.17011","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:b06be33587a43e97f91a627c20a0ac7e51f82e814363838da05eef3632463601","observation_id":"08c7c79b-2c56-4764-b50f-6073c683c43e","resolution":{"observed_at":"2026-08-06T23:50:23.445973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.15658","last_updated":"2024-11-25T17:14:20Z","snapshot_observed_at":"2026-08-16T13:49:41.910701Z","submitted_at":"2024-05-24T15:53:59Z","title":"CoHD: A Counting-Aware Hierarchical Decoding Framework for Generalized Referring Expression Segmentation","version":2},"cited_work":{"arxiv_id":"2405.15658","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.15658","snapshot_observed_at":"2026-08-06T23:50:23.690771Z","title":"CoHD: A Counting-Aware Hierarchical Decoding Framework for Generalized Referring Expression Segmentation","venue":"cs.CV","work_id":"a0781c56-5aff-49cb-a46c-d5cfd7b8a9ac","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.450624Z"},"links":{"cited_paper":"/paper/2405.15658","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:c601aaa889a1106a17076358d5051009c95c15b573d550b51320d3896fb63374","observation_id":"3ba8fd79-3fbd-4731-bda1-b25a15a2fbd1","resolution":{"observed_at":"2026-08-06T23:50:23.696452Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.289079Z","title":"Matrix analysis and applied linear algebra","venue":null,"work_id":"734db809-9369-4e2d-85b9-22a4af48d522","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.456174Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:08bf17b4f9c00889fc4e10daa1728e83e867daaa40b4809c7ba01c58d4b53781","observation_id":"881002cc-da08-4e94-9b36-86b82af250f6","resolution":{"observed_at":"2026-08-06T23:50:24.294036Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.274052Z","title":null,"venue":null,"work_id":"d4f48674-d34e-46c5-8a54-561cb5b318cb","year":2014},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.460868Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:88a0084ba7402479c570aad36a6424845af44f08f4d6bee75b77de8327b7dbca","observation_id":"92649031-db64-4c2c-9ab0-735192d697d6","resolution":{"observed_at":"2026-08-06T23:50:24.279433Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.259295Z","title":"Siri: A simple selective retraining mechanism for transformer-based visual grounding","venue":null,"work_id":"d8fe4c4e-c515-44b1-91af-c0d64b814bc4","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.465327Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:3f83fb7dd74567f59af34691a20a1f0b311895c0aed588744d8a39cde1d011b3","observation_id":"72488162-0515-4ac4-886e-9cf16112dca5","resolution":{"observed_at":"2026-08-06T23:50:24.264057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.243614Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"a774d58c-b451-4d72-aa55-7218356204ad","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.470029Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:3f76d0587575c868b6807ee60902c26259711833bcf5ff40589cb2c9165ec939","observation_id":"f6eb4e40-6224-4237-b50d-b3d76dde49b8","resolution":{"observed_at":"2026-08-06T23:50:24.248952Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00603","last_updated":"2024-12-15T12:04:12Z","snapshot_observed_at":"2026-08-16T13:38:26.385072Z","submitted_at":"2024-06-30T06:08:12Z","title":"Hierarchical Memory for Long Video QA","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00603","snapshot_observed_at":"2026-08-06T23:50:23.474555Z","title":"Hierarchical memory for long video qa","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.474555Z"},"links":{"cited_paper":"/paper/2407.00603","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:6992864903aee0a612e4c1afa779cc587faccb01c6dcc2ec4821ade9a9e7f70d","observation_id":"ccf2a863-c91f-49d6-a381-782b87cdf801","resolution":{"observed_at":"2026-08-06T23:50:23.474555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.228712Z","title":"Uni-adafocus: Spatial- temporal dynamic computation for video recognition","venue":null,"work_id":"593da257-482e-49e0-b78a-04eb89102515","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.479341Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:b6b043b677b435c49c3dbe5013efb1546899cc71136f6921af9463a419a97945","observation_id":"191b1b35-0da6-4e33-8a0d-247b3a060ea4","resolution":{"observed_at":"2026-08-06T23:50:24.233510Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.212651Z","title":"Iterprime: Zero-shot referring image segmen- tation with iterative grad-cam refinement and primary word emphasis","venue":null,"work_id":"1c81bdef-808e-4b8a-a91c-2b5baa947f05","year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.484430Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:129caec93632c68cf36ddf7997fb40bbe101f8052c3c549f570eb161d73d8d57","observation_id":"8be1fb6b-ddb6-4466-92f6-3e7cc8b144e0","resolution":{"observed_at":"2026-08-06T23:50:24.218075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.197985Z","title":"Sam2-love: Segment anything model 2 in language- aided audio-visual scenes","venue":null,"work_id":"0c8d5337-61f4-4f56-9da0-c7032b2e5b7f","year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.489538Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:dfe97ad2c2846b4719f020d6d10bbc70977ceab302205153617af6d02efeebe3","observation_id":"cb973aa9-23da-4116-afe2-42bef671ea2b","resolution":{"observed_at":"2026-08-06T23:50:24.202554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17606","last_updated":"2024-12-02T03:19:04Z","snapshot_observed_at":"2026-08-13T21:35:43.932778Z","submitted_at":"2024-11-26T17:18:20Z","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.17606","snapshot_observed_at":"2026-08-06T23:50:23.494720Z","title":"Hyperseg: Towards univer- sal visual segmentation with large language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.494720Z"},"links":{"cited_paper":"/paper/2411.17606","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:9830cf61182ee63b429ca6b4de8bef591bac1dbe6f6e73925597ef3cac446df2","observation_id":"494fb9ba-fec2-4b55-8b91-96c449cbf13d","resolution":{"observed_at":"2026-08-06T23:50:23.494720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14006","last_updated":"2024-12-18T16:20:40Z","snapshot_observed_at":"2026-08-15T07:01:26.380594Z","submitted_at":"2024-12-18T16:20:40Z","title":"InstructSeg: Unifying Instructed Visual Segmentation with Multi-modal Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14006","snapshot_observed_at":"2026-08-06T23:50:23.499385Z","title":"Instructseg: Unifying instructed visual segmentation with multi-modal large lan- guage models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.499385Z"},"links":{"cited_paper":"/paper/2412.14006","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:5cd142c9f1d49168eafcc4aa4ad91d48c17f5535e9a206af284a0c9dd7d73624","observation_id":"0cb8fa4d-6035-476f-ae78-1a5d6d762ba8","resolution":{"observed_at":"2026-08-06T23:50:23.499385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.182017Z","title":"A large-scale benchmark for food im- age segmentation","venue":null,"work_id":"b9134be0-e140-4828-af8a-52ca84c4e97d","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.504013Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:0d9d936606f7e741d439f451c34fc720dec62b75136d68d1567e35cf8af5dfa5","observation_id":"6c4d208a-bab4-40b6-90d7-116debeb9aed","resolution":{"observed_at":"2026-08-06T23:50:24.187846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.165622Z","title":"Detectron2","venue":null,"work_id":"4cd24c6a-087e-4be6-96a7-3a8f1a66c367","year":2019},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.508334Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:255a7b761481480d1ee76a9280964e7a8820bc9476f8c6a6d3f6a77f525f2b91","observation_id":"e17d219c-1cdf-4ab9-a8b5-4882523f970d","resolution":{"observed_at":"2026-08-06T23:50:24.170493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.150716Z","title":"Semantic projection network for zero- and few-label semantic segmentation","venue":null,"work_id":"47ab7dd9-8d6d-4b92-ac13-01e0336e758e","year":2019},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.512584Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:00c13e6a0c2b707181d8526625393ab0d2810f72aef833c77000a0b9027c668f","observation_id":"6f439054-5bbb-428e-959b-d363152575d9","resolution":{"observed_at":"2026-08-06T23:50:24.155465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.134673Z","title":"Bridging the gap: A unified video comprehension framework for moment retrieval and highlight detection","venue":null,"work_id":"f003df54-9ad4-4f7e-95db-32a7d7ee7492","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.516744Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:1b9c889e4c11cc66c52724cb1a8c376debc78dd08944072af37bd533b2593e86","observation_id":"8541a39c-81cf-4695-a0d3-dce7b50db631","resolution":{"observed_at":"2026-08-06T23:50:24.140251Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.119024Z","title":"Sed: A simple encoder-decoder for open- vocabulary semantic segmentation","venue":null,"work_id":"d9557ef1-89dd-45cd-8c03-ba970511aa9c","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.521827Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:7d55bb669e5fc132ea87cf1d3ee7d2932d7fbfa2009e1b6db188fc311674cbcb","observation_id":"a8f54fe3-8ea2-4cab-997b-ab0e6d579725","resolution":{"observed_at":"2026-08-06T23:50:24.124526Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.105131Z","title":"Alvarez, and Ping Luo","venue":null,"work_id":"e184d97d-0173-4501-b226-0454e19223ac","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.526934Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:7a9ca4c6d9f9619850785c8752fb9340a273db1cad81302d6ebbf6e2334cf931","observation_id":"fe3547b4-b4dd-4f37-83c1-97cfd2ad475c","resolution":{"observed_at":"2026-08-06T23:50:24.109620Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.088706Z","title":"Open-vocabulary panop- tic segmentation with text-to-image diffusion models","venue":null,"work_id":"cbbf586e-c63a-48a2-9c1e-a46203834090","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.531413Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:73e1ea0b12f1741e700f107da41612f59a6bf68309cc91f51b8c39c8e3eb11e9","observation_id":"fe505afc-f573-406b-8518-9bbc306c6719","resolution":{"observed_at":"2026-08-06T23:50:24.094278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.14757","last_updated":"2022-12-29T16:36:55Z","snapshot_observed_at":"2026-08-06T08:59:22.727151Z","submitted_at":"2021-12-29T18:56:18Z","title":"A Simple Baseline for Open-Vocabulary Semantic Segmentation with Pre-trained Vision-language Model","version":2},"cited_work":{"arxiv_id":"2112.14757","doi":null,"metadata_source":"pith","pith_arxiv_id":"2112.14757","snapshot_observed_at":"2026-08-06T23:50:23.619666Z","title":"A Simple Baseline for Open-Vocabulary Semantic Segmentation with Pre-trained Vision-language Model","venue":"cs.CV","work_id":"ec5226e2-f2be-4f97-b540-beabba9fed52","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.536291Z"},"links":{"cited_paper":"/paper/2112.14757","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:0d07f061fd57d542e174d637e751c7e84bfbc8dbe1f580d79d6b1f1162e01366","observation_id":"17813763-b4c7-4cc1-a2ea-c59571aa80ca","resolution":{"observed_at":"2026-08-06T23:50:23.626640Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.072226Z","title":"Side adapter network for open-vocabulary semantic segmentation","venue":null,"work_id":"fd311380-a784-46ec-84d8-4ceb2f3cff08","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.541877Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:8263cb3ee833b456f2df8efb9b16a463cad109f2439d1724c2496ed73914608f","observation_id":"cd68d2cc-bf5d-4dde-bcd9-7e618daee1d8","resolution":{"observed_at":"2026-08-06T23:50:24.077918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.056926Z","title":"Masq- clip for open-vocabulary universal image segmentation","venue":null,"work_id":"5fb193d3-438d-4df5-9008-ed0274d5c8d0","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.546341Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:fafe67f81431bb5e982f75436511421472765d7b0d945fa693dd770a05b8fdc4","observation_id":"d0e369b9-683d-46a4-8293-042169a04220","resolution":{"observed_at":"2026-08-06T23:50:24.061442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.040567Z","title":"Convolutions die hard: Open-vocabulary seg- mentation with single frozen convolutional clip","venue":null,"work_id":"e4128890-7dde-49ad-ad0f-ccde289b506f","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.550790Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:b5421adfcf8b445bd94caef0415d86d51e4ed1df482a01725ceef1818da6d1b0","observation_id":"bd2b2bcc-c07e-4916-84ae-123c09abd7b5","resolution":{"observed_at":"2026-08-06T23:50:24.046307Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.018215Z","title":"Prototypical matching and open set rejection for zero-shot semantic segmentation","venue":null,"work_id":"295f824b-d1c0-43bf-a773-9846266de0c5","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.555457Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:aab2d52869f0714c02217d2eeacf98a2be7bfc430eefedc5d6e775b6b66318ee","observation_id":"0e8d8dd4-b8d1-4016-b6c3-8bb89b40456c","resolution":{"observed_at":"2026-08-06T23:50:24.024982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-16T13:43:48.910620Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-06T23:50:23.560907Z","title":"Flash-vstream: Memory- based real-time understanding for long video streams","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.560907Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:c734d4dd85ac1f593fa29b9715b8a0c9b2531e85da363f1eac1b78e28fee1cd1","observation_id":"fe2be44c-1e16-44a6-ad2d-f8e543fb0667","resolution":{"observed_at":"2026-08-06T23:50:23.560907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:23.993862Z","title":"Scene parsing through ADE20K dataset","venue":null,"work_id":"5678b5eb-6a93-46c1-a5e9-e20928464ce5","year":2017},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.565560Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:6f37ff7e98ac21047b32a393512b20b3592a5ac84eaaf652b40886a2cb194611","observation_id":"d2d361ab-ea73-4d7a-9d59-648b450a27b6","resolution":{"observed_at":"2026-08-06T23:50:24.002543Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-13T05:12:42.667005Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation"},"reference_resolution":{"displayed":66,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":3,"verified_fuzzy":43},"total_outbound_references":66},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 66 of 66 outbound references and 0 inbound Pith citation observations for arXiv:2506.16058."}