{"as_of":"2026-08-10T10:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:244c4579190e6c4af129dd22eb9a969bdb1311e33f845c850ec54412d2b8f1b0","coverage":[{"denominator":116,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T20:10:19.724032Z","state":"measured"},{"denominator":104,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":104,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T13:42:25.796548Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-29T13:33:28.299770Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"cited_work":{"arxiv_id":"2508.11256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.11256","snapshot_observed_at":"2026-06-29T13:33:28.299770Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","venue":null,"work_id":"25489e52-69ed-496e-9a89-e96910750a83","year":2025},"citing_paper":{"arxiv_id":"2604.17914","last_updated":"2026-04-20T07:47:51Z","snapshot_observed_at":"2026-08-02T06:34:24.155889Z","submitted_at":"2026-04-20T07:47:51Z","title":"Beyond Binary Contrast: Modeling Continuous Skeleton Action Spaces with Transitional Anchors","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-10T04:32:04.547197Z"},"links":{"cited_paper":"/paper/2508.11256","citing_paper":"/paper/2604.17914"},"observation_digest":"sha256:8dc9066a58c1090aa27e0d88fa52f40d38027371facc73097af51414063c4b78","observation_id":"f308d82d-f00e-4cfe-8560-93a223e5e313","resolution":{"observed_at":"2026-05-11T11:51:03.571495Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"cited_work":{"arxiv_id":"2508.11256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.11256","snapshot_observed_at":"2026-06-29T13:33:28.299770Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","venue":null,"work_id":"25489e52-69ed-496e-9a89-e96910750a83","year":2025},"citing_paper":{"arxiv_id":"2605.17583","last_updated":"2026-05-14T13:31:23Z","snapshot_observed_at":"2026-08-02T22:28:42.271827Z","submitted_at":"2026-05-14T13:31:23Z","title":"AgentSteerTTS: A Multi-Agent Closed-Loop Framework for Composite-Instruction Text-to-Speech","version":1},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-05-20T21:14:58.814362Z"},"links":{"cited_paper":"/paper/2508.11256","citing_paper":"/paper/2605.17583"},"observation_digest":"sha256:25880ccec610f95bcc52427ab732529a6394438121e620d98464cca747490473","observation_id":"d2be249b-5a9f-4509-b7a5-0bb8fe90acfc","resolution":{"observed_at":"2026-05-20T21:19:03.217532Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"cited_work":{"arxiv_id":"2508.11256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.11256","snapshot_observed_at":"2026-06-29T13:33:28.299770Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","venue":null,"work_id":"25489e52-69ed-496e-9a89-e96910750a83","year":2025},"citing_paper":{"arxiv_id":"2606.26120","last_updated":"2026-05-27T02:47:50Z","snapshot_observed_at":"2026-07-07T00:00:31.981360Z","submitted_at":"2026-05-27T02:47:50Z","title":"Dynamic-dLLM: Dynamic Cache-Budget and Adaptive Parallel Decoding for Training-Free Acceleration of Diffusion LLM","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-29T13:27:45.796650Z"},"links":{"cited_paper":"/paper/2508.11256","citing_paper":"/paper/2606.26120"},"observation_digest":"sha256:f9ab7a4d1e39691b0f771a38bd1ff9e42b547659bca6daa996dcf993bfe3ce0c","observation_id":"0d581b57-d70b-4243-b350-5a8155b7ea8f","resolution":{"observed_at":"2026-06-29T13:33:28.301224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.11256","snapshot_observed_at":"2026-08-02T13:42:25.796548Z","title":"Declip: Decoupled learning for open-vocabulary dense perception","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.20467","last_updated":"2026-05-19T06:27:58Z","snapshot_observed_at":"2026-08-08T06:33:24.325410Z","submitted_at":"2026-05-19T06:27:58Z","title":"DC-Leap: Training-Free Acceleration of dLLMs via Draft-Guided Contiguous Leaping Decoding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T13:42:25.796548Z"},"links":{"cited_paper":"/paper/2508.11256","citing_paper":"/paper/2607.20467"},"observation_digest":"sha256:1559867be257ae7a730ad235660e9104b0e40470457532ad398b06a6742eed22","observation_id":"2648fb87-7a24-4a22-bfd8-9cbc79320386","resolution":{"observed_at":"2026-08-02T13:42:25.796548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2508.11256/citation-record","integrity":"/paper/2508.11256/integrity","json":"/paper/2508.11256/citation-record.json","paper":"/paper/2508.11256"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:09.840942Z","title":"Faster r-cnn: Towards real-time object detection with region proposal networks,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:09.840942Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:9973400ff6c7bcf9ba932f0ac7c43799117a56f21b159301d0bb27e922db4838","observation_id":"b6e465b7-0a4d-4f1e-99af-92c722ea4481","resolution":{"observed_at":"2026-08-05T20:10:09.840942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:09.985717Z","title":"DAB-DETR: Dynamic anchor boxes are better queries for DETR,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:09.985717Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:f8022062d6632b3e268b008a6b1488fa025cb6cd1b7f5f85b91446ec8eff047a","observation_id":"b13cb74d-faf2-41ad-aa96-8b19e342e8fa","resolution":{"observed_at":"2026-08-05T20:10:09.985717Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:10.172898Z","title":"U-net: Convolutional net- works for biomedical image segmentation,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:10.172898Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:69b18cc531178e1da83b09fa357f672ebbf6ec6bef0964ab30e49283ecbff95b","observation_id":"a81e8b4b-1baf-4134-a92b-8061e6e25e0a","resolution":{"observed_at":"2026-08-05T20:10:10.172898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:10.358448Z","title":"Masked-attention mask transformer for universal image segmenta- tion,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:10.358448Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:f70da9cb338665031cf0f4fc95f3a530347d8663c6fba3c4916d79849bfba438","observation_id":"676ab988-432f-499b-bcb3-7877e37249b7","resolution":{"observed_at":"2026-08-05T20:10:10.358448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:10.521983Z","title":"Mask dino: Towards a unified transformer-based framework for object detection and segmentation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:10.521983Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:27567092c3c38f02c85d2280d17d8d8a69df423cad9047f6d4a5c96b4c5cd832","observation_id":"26dd1416-6cde-424a-8587-73b8f536edc4","resolution":{"observed_at":"2026-08-05T20:10:10.521983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:10.665833Z","title":"Enhanced training of query-based object detection via selective query recollection,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:10.665833Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:705be1ced2842b480e39a11988f70f2ef057540425fe24635d8725371ef693b8","observation_id":"3720bf71-4fe1-469c-8eca-f51954994a68","resolution":{"observed_at":"2026-08-05T20:10:10.665833Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:10.784765Z","title":"Deformable detr: Deformable transformers for end-to-end object detection,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:10.784765Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:42d74951c19abd17360dd3e8899256b028fd25d5d2ede0ac1b8c9f3f162d2bf6","observation_id":"fee31577-53ec-4b70-972d-1b8c35b02531","resolution":{"observed_at":"2026-08-05T20:10:10.784765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:10.884468Z","title":"Open-vocabulary object detection using captions,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:10.884468Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:47e24b89220826614978f81235457ed5a18dd1040535b05be7ee06bb74f0b135","observation_id":"a55b8869-a91b-4a9c-9e25-458a2551f58c","resolution":{"observed_at":"2026-08-05T20:10:10.884468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:10.958133Z","title":"Aligning bag of regions for open-vocabulary object detection,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:10.958133Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:1802b61bbb9096973f2ed0a2f4304a96f1ab0838a27f4e0316cdbd62d3fad83c","observation_id":"41ae203e-0c3f-479d-86a3-f58951966f6f","resolution":{"observed_at":"2026-08-05T20:10:10.958133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:11.041423Z","title":"Cora: Adapting clip for open- vocabulary detection with region prompting and anchor pre-matching,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:11.041423Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:3efb9eb1a56d42f0e09fdd8e7605d24e34267ed3bd4d4d0d38239088d434314e","observation_id":"80667852-540e-44d3-9631-282ed239b717","resolution":{"observed_at":"2026-08-05T20:10:11.041423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:11.095095Z","title":"Cat- seg: Cost aggregation for open-vocabulary semantic segmentation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:11.095095Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:c0bd71d84ad1febb1c649aad417790c4d6b400dffd0bf6e73c77a2f46d170044","observation_id":"4b505aa2-f179-4821-b956-5f601838d1c6","resolution":{"observed_at":"2026-08-05T20:10:11.095095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:11.167225Z","title":"Learning transferable visual models from natural language supervi- sion,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:11.167225Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:66207fac8a03b45bc31a5800b6ddd1fbe4ad608b62057a077539c86814352566","observation_id":"005594f2-0264-46a5-acf0-9f44c9ce3e10","resolution":{"observed_at":"2026-08-05T20:10:11.167225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:11.249205Z","title":"Scaling language- image pre-training via masking,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:11.249205Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:1538a131c2d399a9afe14a8748c0c704855921e0affd4740cc4e28e50a52144e","observation_id":"254a089b-43b8-4ebb-bb04-7832a204a03b","resolution":{"observed_at":"2026-08-05T20:10:11.249205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:11.367823Z","title":"Clim: Contrastive language-image mosaic for region representation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:11.367823Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:2208b696ef38e34a0758ede5cdaab40c9626017f44829fa97651ef23f32cd025","observation_id":"51e3ac40-5b5c-4e61-b2e7-35fc905e8009","resolution":{"observed_at":"2026-08-05T20:10:11.367823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:11.471990Z","title":"CLIPSelf: Vision transformer distills itself for open-vocabulary dense prediction,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:11.471990Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:ad0bf165bff21e3fc228be67988277c59f7fab7936bcd87184d03dd3211e40e3","observation_id":"3cbf97b9-dfb9-46f1-b7aa-f2fa527463e9","resolution":{"observed_at":"2026-08-05T20:10:11.471990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:11.550247Z","title":"Regionclip: Region-based language- image pretraining,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:11.550247Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:4bd21400d132ac6c3b731dba150543762ac38eda5de0a98329d30f5bf201bbff","observation_id":"4ffefd41-6b38-4499-b361-7086adddb965","resolution":{"observed_at":"2026-08-05T20:10:11.550247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:11.632230Z","title":"Ov-dquo: Open-vocabulary detr with denoising text query train- ing and open-world unknown objects supervision,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:11.632230Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:0fd8aa6f59b5ccce4f9fec431fc0a66298cbd57de8dbad56b16e446228231a90","observation_id":"d66d7f3c-8cb1-4afd-8f2a-3fd23eae852f","resolution":{"observed_at":"2026-08-05T20:10:11.632230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:11.712835Z","title":"Open-vocabulary object de- tection via vision and language knowledge distillation,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:11.712835Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:ba4d5192332b41ee3cbf5ec0151cfea4517e7ca988a5202a6aac9f5147e88365","observation_id":"92b6f947-9229-436a-9b7a-8db1109c8f04","resolution":{"observed_at":"2026-08-05T20:10:11.712835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:11.778514Z","title":"F-vlm: Open-vocabulary object detection upon frozen vision and language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:11.778514Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:967da8a812402e36506c87aba918abfe69440ac2a542a40b7cb8393ff5e1df10","observation_id":"b511101e-e005-4623-89cc-81a6ce653cf9","resolution":{"observed_at":"2026-08-05T20:10:11.778514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:11.880929Z","title":"Open-vocabulary semantic segmentation with mask-adapted clip,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:11.880929Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:d94c9598d71ded873b9cea5e1d4863406990ba37008bc644975d85f92569ab6b","observation_id":"59667dd5-5997-409f-ad3a-25b3c4c980f5","resolution":{"observed_at":"2026-08-05T20:10:11.880929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:12.005807Z","title":"High-resolution image synthesis with latent diffusion models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:12.005807Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:19d79090f3052b1d19157003ee6ec68b8d30fb83ac6dd126a863c788b554b0a8","observation_id":"c4ab2810-05bc-487f-9d45-591ebcfe8151","resolution":{"observed_at":"2026-08-05T20:10:12.005807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:12.093572Z","title":"A survey on open-vocabulary detection and seg- mentation: Past, present, and future,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:12.093572Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:fa929120f6551032c13e62cd48b7b7ab33303b66cc908185ab79bcf42d9c7a93","observation_id":"3cd92037-2c8a-4627-b85f-0d5216336030","resolution":{"observed_at":"2026-08-05T20:10:12.093572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:12.169894Z","title":"Towards open vocabulary learning: A survey,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:12.169894Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:4647c0917fe980ca9adb359be21d5b98c832f08221d8f1c499668f0ddfa1c79b","observation_id":"6ebec5cc-a8ed-426f-8684-035a8e05981c","resolution":{"observed_at":"2026-08-05T20:10:12.169894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:12.210528Z","title":"Object-aware distillation pyramid for open-vocabulary object detection,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:12.210528Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:96c1c02cddbe3e82c7c5301e3b15af88b3adec4e196473a1b48a1754425501e2","observation_id":"b9605788-1ab7-4438-beb8-d969d2a32bbe","resolution":{"observed_at":"2026-08-05T20:10:12.210528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:12.287880Z","title":"Groupvit: Semantic segmentation emerges from text supervision,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:12.287880Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:54c374bc78e000cf5dabb950cd0bacc254b7fbad537c2c81debe5886e915e527","observation_id":"23979c0f-7801-4bfa-aca1-778effe26441","resolution":{"observed_at":"2026-08-05T20:10:12.287880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:12.348029Z","title":"Scaling open-vocabulary image segmentation with image-level labels,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:12.348029Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:d115af48a408162ef77f8f39b70bd1a8c71cc3c00549ecdc29bf8b39538ba7ca","observation_id":"8a0721e4-2a9c-4870-9444-f119a472b4eb","resolution":{"observed_at":"2026-08-05T20:10:12.348029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:12.474425Z","title":"Taming self-training for open-vocabulary object detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:12.474425Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:705486be3f98d68eb674c224709bc997011dcffb11cc10161d70ef03b95e5a09","observation_id":"b36b47c6-cb06-4053-a69a-b1310328eeb5","resolution":{"observed_at":"2026-08-05T20:10:12.474425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:12.594935Z","title":"Detecting twenty-thousand classes using image-level supervision,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:12.594935Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:812ab129410385684d8ef7b5233b486ca5f45fa866ddacb3197c332f30f25eaa","observation_id":"54c81a43-db65-43fc-9ea9-fe6458e595fe","resolution":{"observed_at":"2026-08-05T20:10:12.594935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:12.694061Z","title":"Open-vocabulary detr with conditional matching,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:12.694061Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:5883e03ded5a7b67bfd12bf7008a7f4724f938d8031e8f07afbe55d58a9bbf8a","observation_id":"87e73afb-085b-43e8-8cb6-617f97ab90ce","resolution":{"observed_at":"2026-08-05T20:10:12.694061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:12.791515Z","title":"Global knowledge calibration for fast open-vocabulary segmentation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:12.791515Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:85bab7ae49720f462c52317e13ac2f7ecda348e510ed73595618d12fe8fd53b1","observation_id":"7933283e-d56e-48c3-8910-4478e6c73780","resolution":{"observed_at":"2026-08-05T20:10:12.791515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:12.869705Z","title":"Densegrounding: Improving dense language- vision semantics for ego-centric 3d visual grounding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:12.869705Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:939021b99dda23f59d7095a6a224829c6ad42bb977f6c8c85210da24c55cd0b9","observation_id":"70b6f036-7c3c-46b1-9a72-94c8db72d67a","resolution":{"observed_at":"2026-08-05T20:10:12.869705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:13.005986Z","title":"Detect anything 3d in the wild,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:13.005986Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:eecff95fcfb37b65b29dc2a606726431be7e2c7575840e95563f3875af125bce","observation_id":"65466bf6-0e39-45c0-bee0-9607668b5824","resolution":{"observed_at":"2026-08-05T20:10:13.005986Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:13.097939Z","title":"Sam3d: Segment anything in 3d scenes,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:13.097939Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:159eeb03777dc3c72c63a887c1052e92d91c04173b43e9ecbe4cadf6b8fd31af","observation_id":"9e3e68d5-8023-4b95-8956-d49f03946f03","resolution":{"observed_at":"2026-08-05T20:10:13.097939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:13.236236Z","title":"Ovir-3d: Open-vocabulary 3d instance retrieval without training on 3d data,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:13.236236Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:8fa5aebcb7a47cd8d1c732aa740ab4ae5be91d149ebe1f92bcaa2ccfa492059c","observation_id":"a294c18a-4d55-4924-82c5-b2ba4e5da431","resolution":{"observed_at":"2026-08-05T20:10:13.236236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:13.309695Z","title":"Open- vocabulary object 6d pose estimation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:13.309695Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:36b66c14c329791cd13cd2ca0723e19aaca62001b4eceac6866969b04abf8d7c","observation_id":"f63594e5-8b8c-45dd-82cd-bcd277462492","resolution":{"observed_at":"2026-08-05T20:10:13.309695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:13.348117Z","title":"Open3dis: Open-vocabulary 3d instance segmentation with 2d mask guidance,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:13.348117Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:63459f2bd7ee1791a5013f6bdb72dccc1b04ce2872d19284d4c23523ba68c2a4","observation_id":"c5725e7e-534b-46ba-b95f-7ec2ef16536b","resolution":{"observed_at":"2026-08-05T20:10:13.348117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:24.100695Z","title":"Openmask3d: Open-vocabulary 3d instance segmenta- tion,","venue":null,"work_id":"bdc06b72-5e2d-47f9-81ca-fdad27fd1a32","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:13.444699Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:fdcfb713246a55f1222c905712ca6476afb16806e65098e18b67ccfda391d931","observation_id":"97ff88ea-e009-408f-9049-a35ac8abe379","resolution":{"observed_at":"2026-08-05T20:10:24.104216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:24.091865Z","title":"Clip-vis: Adapting clip for open-vocabulary video instance segmentation,","venue":null,"work_id":"634667b5-4525-486e-a6b7-6bc8dbdc2984","year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:13.562789Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:d5aa7830fa2796efbbcef67e6f71445fb9386f544189d4762bd6ab86f62a88e1","observation_id":"a6081716-0575-49b1-b84c-03f10a955575","resolution":{"observed_at":"2026-08-05T20:10:24.095328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:24.083207Z","title":"Semantic and sequential alignment for referring video object segmentation,","venue":null,"work_id":"c3674ca3-34f1-4f92-87d2-2f3746e24768","year":2025},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:13.698246Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:8aa2be1a39eecf27d0c81b4da5ebf8d39a55e8486990a7cd69d4c8440e5146e5","observation_id":"ccb798fd-ca9f-45e8-9765-4fb34d4485ae","resolution":{"observed_at":"2026-08-05T20:10:24.086407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:24.073372Z","title":"Unified embedding align- ment for open-vocabulary video instance segmentation,","venue":null,"work_id":"222a00eb-7aec-49ff-8e64-08d2c4abd13a","year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:13.776707Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:b58a4081138af37315d741468c432c57a7e0a67c826f879fa45a63d3848ca9d5","observation_id":"a6de5595-2620-4bed-8115-fd83c51a5ce7","resolution":{"observed_at":"2026-08-05T20:10:24.076767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:24.063767Z","title":"Sigmoid loss for language image pre-training,","venue":null,"work_id":"c4f7f652-9536-4f20-bf57-799c4ff6b217","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:13.860179Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:d57de3ef54d2c1eeb109bcb63cfe46e631b12f818ebb9c765f8c4bc1a0e639c0","observation_id":"58b3e3ad-cc6a-4666-a1a0-406302272a52","resolution":{"observed_at":"2026-08-05T20:10:24.066746Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:24.054295Z","title":"Learning mask-aware clip representations for zero-shot segmentation,","venue":null,"work_id":"60ffec4f-d67c-4f14-8a4c-7c17db23d62c","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:13.940164Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:06383ce8d18a9cace9ef9c5959e018505385113d7048b9acb1b2cc32facfb9ea","observation_id":"a556ba8d-5184-490a-9d16-65668d7f8f84","resolution":{"observed_at":"2026-08-05T20:10:24.057529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:24.044765Z","title":"Language-driven semantic segmentation,","venue":null,"work_id":"03f4a023-da10-447b-936d-2f5b50a95dce","year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:14.042602Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:7e1d5450896ef919de6de153f59e2db9a2f1cb04c7891bb7b202573f2166db4a","observation_id":"cadb0cd1-c63c-48b8-be3f-afa148e45b02","resolution":{"observed_at":"2026-08-05T20:10:24.048518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:24.035277Z","title":"Open vocabulary semantic segmentation with patch aligned contrastive learning,","venue":null,"work_id":"a9f8f73a-6764-4040-9cd6-31ef968942e2","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:14.204613Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:0ffb44cd8a530980bb02b1fe780250663e33a3c856139959e39e3ffab6864321","observation_id":"65b5c39a-8931-4ff1-bd2b-2c2ff2964766","resolution":{"observed_at":"2026-08-05T20:10:24.039381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:24.024530Z","title":"Sam-clip: Merging vision foundation models towards semantic and spatial under- standing,","venue":null,"work_id":"22cb830f-fc3f-426f-9287-361d2e1abc49","year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:14.294410Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:52641985adaa0415b654f416f8e62076afb3b0cee2b961788ec1817b286959de","observation_id":"b25a119c-0a21-4bb7-8a81-19d80f4db41c","resolution":{"observed_at":"2026-08-05T20:10:24.028511Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:24.015905Z","title":"Open- vocabulary sam: Segment and recognize twenty-thousand classes in- teractively,","venue":null,"work_id":"6b063425-a5e4-4481-8418-2746f1040814","year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:14.362385Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:28ee91e41dd31530229ce571646774e38c2eb25e95f6fd32b1b4c04f5dde047e","observation_id":"c51eff8f-92a3-4ad1-a9d2-80ea322a2df9","resolution":{"observed_at":"2026-08-05T20:10:24.018831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:24.005260Z","title":"Frozenseg: Harmo- nizing frozen foundation models for open-vocabulary segmentation,","venue":null,"work_id":"7e8ee232-64fd-4f9a-a5fd-3bfe86ba7f18","year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:14.437271Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:ec67ad30505f7ea2554764f698b81c14c3dc25a16b8a2aa3b628ecd732da91b8","observation_id":"61478947-f6cf-4a1a-a6fe-e8d95eb959e3","resolution":{"observed_at":"2026-08-05T20:10:24.009620Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.994186Z","title":"Segment anything,","venue":null,"work_id":"ac405997-c448-45ff-b2d5-5833ff2bfc77","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:14.544984Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:89b4f3e729706a6772ee5165526297bda4179891338bf8fd50291cdd856b17bb","observation_id":"aee0f0a6-dac6-4a7c-94a9-c223bfc651c2","resolution":{"observed_at":"2026-08-05T20:10:23.998121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.983527Z","title":"Rep- resentation alignment for generation: Training diffusion transformers is easier than you think,","venue":null,"work_id":"5789a209-6bb9-4ac7-8189-acc95bf9618d","year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:14.679168Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:e046c6c82289fddb5d605e67196b3a8055f70c64d07e1a69c0ae1f4f44997da7","observation_id":"f0d793ae-1c9d-4c32-b3eb-7c4d95387681","resolution":{"observed_at":"2026-08-05T20:10:23.987948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.972364Z","title":"A convnet for the 2020s,","venue":null,"work_id":"0057d1a7-0b5f-49e3-a14e-f673ee65c631","year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:14.803588Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:4334f977f28a81bf41620db3a8c210f7f156b412eed8d01ba848ddfa3784da24","observation_id":"986b009c-be4d-49bf-b945-6923fd7e86fc","resolution":{"observed_at":"2026-08-05T20:10:23.976472Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.961903Z","title":"Deep residual learning for image recognition,","venue":null,"work_id":"88117188-c844-4208-b519-3297f0e526e1","year":2016},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:14.906924Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:6ce0ffabe9603d771fd02ff2fc241ceea76c8090fb14e82ddacdb3b2cc40d323","observation_id":"6619ccf9-cac6-4c3b-a27e-5dcf43a1f0f7","resolution":{"observed_at":"2026-08-05T20:10:23.965878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.951965Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale,","venue":null,"work_id":"62d4a44c-cd91-4ca3-8c70-920bebb3f2c8","year":2020},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:14.996849Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:3100bf195948a3ef5da09ae47f691c5638de96ee1b74ea96bfc4efb2933e6924","observation_id":"9dac5372-388a-45fd-83ee-fafa1ae2fb26","resolution":{"observed_at":"2026-08-05T20:10:23.956105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.941208Z","title":"Attention is all you need,","venue":null,"work_id":"37523c7b-21f9-4cc8-9d3c-7ccbbc24d45e","year":2017},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:15.076869Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:9357c6b9e08958ef1a6d2c493589a9048cc7567264f458743b82448763c551b1","observation_id":"cbe289d1-a07c-4257-b58e-5fff0805e3be","resolution":{"observed_at":"2026-08-05T20:10:23.945286Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.932368Z","title":"Emerging properties in self-supervised vision transformers,","venue":null,"work_id":"a3bc41a7-07aa-4e0c-aa43-1719ac21440f","year":2021},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:15.183668Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:ee0e973692c7fe141464953de6a6d3075c03e741c4b378073e828d279f80887b","observation_id":"f708fa24-1712-4b72-a4a0-53afdca8449b","resolution":{"observed_at":"2026-08-05T20:10:23.936097Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.919680Z","title":"Dinov2: Learning robust visual features without supervision,","venue":null,"work_id":"0404d469-491e-491c-b593-a481779d7ff9","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:15.290363Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:9e3544321e8fb01e27d2f7d56ab51c81e525a9d9b8fa18e3882173c693265e99","observation_id":"92be49f7-c9f9-4161-83cc-5667cf040cd3","resolution":{"observed_at":"2026-08-05T20:10:23.924042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.910006Z","title":"Sam 2: Segment anything in images and videos,","venue":null,"work_id":"fca1f0f4-e9a6-4fb5-9ac0-984898626d35","year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:15.436149Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:4f7a178a17c0055472100aae5f679b809c61d7f1b08dbb0a233c3d6c14d6309c","observation_id":"25aedd9a-1bc6-4701-8e1b-018f207f79f8","resolution":{"observed_at":"2026-08-05T20:10:23.913865Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.898512Z","title":"Sclip: Rethinking self-attention for dense vision-language inference,","venue":null,"work_id":"298d2673-d916-4438-94b5-5bf5e2d71081","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:15.507929Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:7ebdc0d59a5ae8622196f7afe2c59064e109743e64976c3be43c591819cea16b","observation_id":"be12a41f-a993-416a-a2ec-6610f7d176a2","resolution":{"observed_at":"2026-08-05T20:10:23.902701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.888401Z","title":"Explore the potential of clip for training-free open vocabulary semantic segmentation,","venue":null,"work_id":"d82fb9b4-539b-4c23-ab57-0481a99992ae","year":2025},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:15.616014Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:ef7261655555845ab41f47141fab31b9d985f608e93675d64e96946c3347925d","observation_id":"bd6458fc-e7a9-4dc9-bd71-68a633858bc7","resolution":{"observed_at":"2026-08-05T20:10:23.891619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.877669Z","title":"Clearclip: Decomposing clip representations for dense vision-language inference,","venue":null,"work_id":"8a14e996-26fc-4c35-8cdb-19b674edbf14","year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:15.780654Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:bf6fc5bac6363a92734672c5cfe85827187d4c03389341cbe2ed0a13cdc0056a","observation_id":"5dd4cad7-b9f6-48e7-9757-c62281e5afcd","resolution":{"observed_at":"2026-08-05T20:10:23.881675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.868681Z","title":"Clip-dinoiser: Teaching clip a few dino tricks,","venue":null,"work_id":"42f04e9a-fcaa-4c77-87a0-94306ca877ef","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:15.922960Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:2a9e6077a0fa22d77cdd6debf4e8d9a60177d4f1a8e0be291a4251e51b882e4f","observation_id":"4b4c6637-59b2-4460-949b-50c2aaf538e0","resolution":{"observed_at":"2026-08-05T20:10:23.871784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.859702Z","title":"Diffusion model is secretly a training-free open vocabulary semantic segmenter,","venue":null,"work_id":"9bd2f1c2-2203-456e-97b2-7fbadbaf213e","year":2025},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:16.113871Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:cd788585910c2ffff00c48785f6c177d156a1dc4afc0710725f7192f5875aa77","observation_id":"41efa2dc-4218-4e97-864a-f07ff8cdb94b","resolution":{"observed_at":"2026-08-05T20:10:23.862851Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.849663Z","title":"Dynamic prompt learning: Addressing cross-attention leakage for text-based image editing,","venue":null,"work_id":"ed44c789-aec2-4616-9bd7-dd8c98ffb249","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:16.260625Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:352d5dee281091877b425f5feb4c31c307cc20899d52872cc8c3f5d618f6eb6b","observation_id":"2ba01bef-113a-4131-ba3e-c5c23fb7c13a","resolution":{"observed_at":"2026-08-05T20:10:23.852800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.839656Z","title":"Cliper: Hierarchically improving spatial representation of clip for open-vocabulary semantic segmentation,","venue":null,"work_id":"149b29cd-3386-4008-b2f4-f01527825fe0","year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:16.380258Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:8428a02c60dd1e968b645c407cbe4c1719a0bd4325ca7e5cc2a906edafa132a2","observation_id":"600b1258-c5b5-49a7-8d24-4f7667c938e0","resolution":{"observed_at":"2026-08-05T20:10:23.843352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.830255Z","title":"Silc: Improving vision language pretraining with self-distillation,","venue":null,"work_id":"126b3f67-9436-4fe6-a50d-524aa4371639","year":2025},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:16.477605Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:977692a7ae9268b31dad5e93e416c600426601f849ad6e9148a5b7f481327451","observation_id":"2f936345-1d95-44fb-b4dd-c5bcd21ee3b3","resolution":{"observed_at":"2026-08-05T20:10:23.834223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.821183Z","title":"Exploring open-vocabulary semantic segmentation from clip vision encoder distillation only,","venue":null,"work_id":"5ef6a41a-c64c-4b16-aeba-e263029c95c8","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:16.681775Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:2f148dd727363682c4864ac630b06c3c580a1ef1c876b72a6b11a27a399a8d75","observation_id":"79b85762-b0b9-45ce-8510-9b50eb619c28","resolution":{"observed_at":"2026-08-05T20:10:23.824247Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.810294Z","title":"Mask r-cnn,","venue":null,"work_id":"915a845b-e92a-4723-aa31-2d7c94e4b438","year":2017},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:16.782295Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:4d6e0b20b0df2b82e499eb55ab2a6b1a6e20e73b6ffb3a250ff1b0290a9d01d0","observation_id":"d5193b3c-ebda-45e2-bbfb-feada88a460f","resolution":{"observed_at":"2026-08-05T20:10:23.814331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.801088Z","title":"Relational knowledge distilla- tion,","venue":null,"work_id":"2a612add-9ad6-45fe-b513-bc6cb0c98a22","year":2019},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:16.829689Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:94ff30f278988497075462d4339763aca3b3c16b6891d304a69d988a24d7ee74","observation_id":"205d5565-0b63-4285-ad22-aede1d0d3094","resolution":{"observed_at":"2026-08-05T20:10:23.804002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.792347Z","title":"Microsoft coco: Common objects in context,","venue":null,"work_id":"a9d0f75e-5138-4f50-b202-b73a6631ec1e","year":2014},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:16.894190Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:d5a7ef0b4db86bcbf8e2f66ac10f55e207c41c2028f8e5406712fe2dd3747dce","observation_id":"5277c74d-472f-4a68-bc85-9d836843269a","resolution":{"observed_at":"2026-08-05T20:10:23.796031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.782038Z","title":"Decoupled weight decay regularization,","venue":null,"work_id":"d0537e8c-8c37-4590-92d8-fd678c3ea043","year":2017},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:16.986846Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:249c6eaef97fe2b25d951fb9fc5320fb9336eb8aa6381479db6524d145fe140a","observation_id":"972f6476-b4b4-464e-84df-73550f4a0523","resolution":{"observed_at":"2026-08-05T20:10:23.785855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.772160Z","title":"Eva-clip: Improved training techniques for clip at scale,","venue":null,"work_id":"03782380-10f2-4890-983b-1ef4d14ad441","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:17.061374Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:5147d6dcac936ad437cf26deb8f62871451e86ffb33f20a14ad426c2bf38e084","observation_id":"c5f7be36-a448-4687-b029-a22148928726","resolution":{"observed_at":"2026-08-05T20:10:23.775193Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.762072Z","title":"Vision transform- ers need registers,","venue":null,"work_id":"c9199cd0-1595-47bd-9071-c063594b144e","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:17.191747Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:150df076bfb9cb8bfb2a3e1a25450dfebf1a4655b8378e30d2bd60d2ec1e78b1","observation_id":"49b7510c-7d13-4880-83a3-33a7159b7b58","resolution":{"observed_at":"2026-08-05T20:10:23.765935Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.752512Z","title":"Language-grounded indoor 3d semantic segmentation in the wild,","venue":null,"work_id":"72f4a630-9a60-40c3-b73c-72815311ed9b","year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:17.276128Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:9e91d93cccd45022712c3e30e080391d43b9f23951e46476f8afd6ed3db40be0","observation_id":"ac1612bb-55dc-4acf-b6c9-47460689030c","resolution":{"observed_at":"2026-08-05T20:10:23.755803Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.742824Z","title":"Isbnet: a 3d point cloud instance segmentation network with instance-aware sampling and box-aware dynamic convolution,","venue":null,"work_id":"460c71a4-2117-41e8-8a05-1caaed962a79","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:17.352021Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:c08215c2ee92fa8edba63bae66e35c1dff0dbde2312dcc50769d00c8619fd6e8","observation_id":"5275d7da-9d4d-4006-9db1-1ea7e7fc4c0a","resolution":{"observed_at":"2026-08-05T20:10:23.746015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.734161Z","title":"Mask3d: Mask transformer for 3d semantic instance segmentation,","venue":null,"work_id":"d32a5f21-ed96-47ea-bdc5-5e60229ea5f2","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:17.411695Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:17f05e60fc4911008b1aa0674194460695e14721bbb5b87b95771d4098d71a49","observation_id":"987cc53e-740e-42e2-923e-af661b8ca986","resolution":{"observed_at":"2026-08-05T20:10:23.736986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.726270Z","title":"Openscene: 3d scene understanding with open vocabularies,","venue":null,"work_id":"c568b679-7794-41f9-97d0-d4399aaee7f8","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:17.473786Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:173fc9f43d4a471511633fdf64635e8a4c8c4a7a59b9d693ef45633f7ace25ec","observation_id":"9e692ed7-6df9-4e5c-81c2-946d6f8602a6","resolution":{"observed_at":"2026-08-05T20:10:23.729040Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.717256Z","title":"A density-based algorithm for discovering clusters in large spatial databases with noise,","venue":null,"work_id":"4cb893bd-7050-42bc-baf6-d188c7b62bb1","year":1996},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:17.540970Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:956c59af040423e4b1050d1631dab19a39a9404d77db1095c4b349d18d72b09e","observation_id":"b2983b14-f286-460e-be1d-1c5dd73729b2","resolution":{"observed_at":"2026-08-05T20:10:23.720545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.709213Z","title":"Openins3d: Snap and lookup for 3d open-vocabulary instance seg- mentation,","venue":null,"work_id":"eedf3c42-e532-4ca6-ace8-8250d68e125e","year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:17.680891Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:72cd94332398ad0d14d06aa2b585f19bbc1a91f24ee6ee401333f8171bf083b0","observation_id":"b13a1cb6-7574-49f7-a680-3c51c19c1489","resolution":{"observed_at":"2026-08-05T20:10:23.711917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.693704Z","title":"In defense of on- line models for video instance segmentation,","venue":null,"work_id":"75bc8cda-11a3-45fb-902b-1c8911d6bd57","year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:17.852582Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:598a9ff8d57b3507081d63f5918809491b9aa9e0d448e98f64c308f6fd9238fe","observation_id":"7de92333-3e91-4dfd-b235-963bf0d22236","resolution":{"observed_at":"2026-08-05T20:10:23.696296Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.685500Z","title":"Simple online and realtime tracking,","venue":null,"work_id":"22fccac2-6bfd-431f-86e5-8f60bf0432b6","year":2016},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:17.953660Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:d48b92ae0cf970fa7a91b2e22c6d310067d19da5d8dcc9490bfda0b17cc9820f","observation_id":"7efe7155-1222-4a61-9e89-ad21db25f6c7","resolution":{"observed_at":"2026-08-05T20:10:23.688453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.676104Z","title":"Opening up open world tracking,","venue":null,"work_id":"054ff3d1-4921-49f0-b78f-cca28b1af6ce","year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.020767Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:01f9a978597ab1ef3a90e87d94d2dcb69009947f2ff5cefa3b7e0557e4cc5e04","observation_id":"f8650d5b-ca4f-4e2c-9d15-bb844a8fbff5","resolution":{"observed_at":"2026-08-05T20:10:23.678693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.668542Z","title":"Xmem: Long-term video object segmentation with an atkinson-shiffrin memory model,","venue":null,"work_id":"e0fe803e-6670-48aa-8941-788251b543b6","year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.097949Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:59c480f5474d732387a068dd5a8bdc445a31d4ab251647f06c55237dc0d1b388","observation_id":"1bda6104-ec21-4a83-bebc-6bacaeb08a6f","resolution":{"observed_at":"2026-08-05T20:10:23.671067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.659593Z","title":"Towards open-vocabulary video instance segmentation,","venue":null,"work_id":"4a26911c-2760-478d-aebf-9f963cb8b919","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.158794Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:46eefd5097b029ef248cc612fc5f39eab607f6e34118fc42ae41d24fc9053304","observation_id":"ad7dbb41-7e45-494e-8aa1-47daaebdad62","resolution":{"observed_at":"2026-08-05T20:10:23.662817Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.701883Z","title":"Video instance segmentation,","venue":null,"work_id":"3c931a19-f741-43c7-abe7-2dd56cd59a6f","year":2019},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.232954Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:469c74abb8857b849a0ca9f7b3ab618f98b62d98f6af01b8bcca67ae3eae3ca6","observation_id":"5e01609f-edcc-41c0-8839-39878b1d24b9","resolution":{"observed_at":"2026-08-05T20:10:23.704463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.652246Z","title":"Lvis: A dataset for large vocabulary instance segmentation,","venue":null,"work_id":"6aadc64a-4f14-47b9-a41c-11708c17e8b6","year":2019},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.300865Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:9f9bd98de4ec4fd027f10db754d395a1eaf0ebfd82cb956318ad8574db728bfa","observation_id":"3c4543e9-4a42-49c0-99cb-81610213c364","resolution":{"observed_at":"2026-08-05T20:10:23.654980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.644318Z","title":"Occluded video instance segmentation: A benchmark,","venue":null,"work_id":"0d9ab62d-4e0b-4b96-92fc-2ba654d47599","year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.362458Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:897ea4f57c30fdcb30d559fcd171d662bb3ac8377d06a9bdf10a8112453badb0","observation_id":"33b1d745-9ae6-4fab-91c6-5b7a5caa95db","resolution":{"observed_at":"2026-08-05T20:10:23.647349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.635855Z","title":"Burst: A benchmark for unifying object recognition, segmentation and tracking in video,","venue":null,"work_id":"e263080d-871d-41c9-924b-485ce76549c1","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.428705Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:4685f3f5f983342d018e51d2af7620ca2b27b98d81678d73e75ec2eb999dae90","observation_id":"897b2fb6-e632-4d5f-aed8-d1bf679cb521","resolution":{"observed_at":"2026-08-05T20:10:23.638704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.626619Z","title":"Fs6d: Few-shot 6d pose estimation of novel objects,","venue":null,"work_id":"decfa737-4e19-4bf2-ba7f-30980970c8e3","year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.508453Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:72fc0aa766727b72069db20ed7cf2af9be8a1211789204619354d2ad281d5192","observation_id":"6f2340bc-7af9-4092-9e04-cac822a5bb02","resolution":{"observed_at":"2026-08-05T20:10:23.629572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.617906Z","title":"Semantically-enriched 3d models for common-sense knowledge,","venue":null,"work_id":"604d856e-556d-40cc-b255-35e77f2e33cd","year":2015},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.606981Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:49e3f5413eac6a4acd6d13b06fdce3d546118997ec99ccb34cd3c81e3cd8002f","observation_id":"11d3065d-c56c-43d4-86b7-dc4ee7ebef54","resolution":{"observed_at":"2026-08-05T20:10:23.621276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.608888Z","title":"Normalized object coordinate space for category-level 6d object pose and size estimation,","venue":null,"work_id":"69f2e521-1bbe-4685-a307-5e53bc8d22c0","year":2019},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.686622Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:b92e10998fd4fcd18fcd1b98bd0d12196b2a232205b4127500d3dc4f61a12534","observation_id":"400933f0-9f5a-40b3-9f94-ab77701c2ef2","resolution":{"observed_at":"2026-08-05T20:10:23.612037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.601228Z","title":"Bop: Benchmark for 6d object pose estimation,","venue":null,"work_id":"fd012114-0515-4dbf-82e6-fc76f9e9477a","year":2018},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.797052Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:38616efe02548d91e9ce16ed1286be73c1fb966d32c1a508385a0285190d8282","observation_id":"aba4668e-f6a4-421e-9de1-5324e94dbd0f","resolution":{"observed_at":"2026-08-05T20:10:23.604203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.594072Z","title":"Bop challenge 2020 on 6d object localization,","venue":null,"work_id":"77f3cc80-9370-417e-8feb-db752c957971","year":2020},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.894916Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:485c5cd5fffb7e9273807f995c37a37a063b2cc4ff656a61e9bc2b335c857908","observation_id":"15346383-ffe8-4f89-990c-41b4c8cb622e","resolution":{"observed_at":"2026-08-05T20:10:23.596699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.586463Z","title":"Swin transformer: Hierarchical vision transformer using shifted win- dows,","venue":null,"work_id":"9991b14a-a0ef-4f08-9243-b233ec076fc6","year":2021},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:18.985082Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:33e5c25705bc72c35ead0d00952e45f59b570baed7064c282fdce470405b9a0c","observation_id":"2c6b7203-98e8-47a2-815b-c7245133b94a","resolution":{"observed_at":"2026-08-05T20:10:23.589444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.577073Z","title":"End-to-end object detection with transformers,","venue":null,"work_id":"2dd82240-4397-44d1-8817-3e60499b4d02","year":2020},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:19.045927Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:d0630d18aaf759c4d9e0572db87b8f0be2aebc3aef3fe9166cfb352f2789bdd0","observation_id":"9fbd2366-3de6-43a1-b8b5-861ae8405fb9","resolution":{"observed_at":"2026-08-05T20:10:23.579542Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.567287Z","title":"Region-aware pretraining for open-vocabulary object detection with vision transformers,","venue":null,"work_id":"f8e9b24a-966a-446b-95f3-cca56b15bb6a","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:19.131544Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:be6f56448a2aefeee9330acfdb1de736434bcbb7f1021d3b5110ef3b97048f02","observation_id":"40a25b6f-db23-4027-955a-9d5abd0cdeea","resolution":{"observed_at":"2026-08-05T20:10:23.570341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.559628Z","title":"Contrastive feature masking open-vocabulary vision trans- former,","venue":null,"work_id":"72d1dd8d-2972-4fa9-993f-90b9c09d5958","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:19.218112Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:fbe4dece8f1ebfee59c12d518bc7dcb62a89b42ab6f7d35eed909c6a1aee6085","observation_id":"2fba74c1-a24c-448e-83bc-cfa1b4c1fc96","resolution":{"observed_at":"2026-08-05T20:10:23.562591Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.544285Z","title":"A simple baseline for open-vocabulary semantic segmentation with pre-trained vision-language model,","venue":null,"work_id":"7d255014-1e44-4119-a271-aa58b838fd85","year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:19.393386Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:546693a51bdb976ff35b1447f4f73a3aea121228dd7897251f0670246161aecb","observation_id":"941618b2-67e7-4535-9552-166819674ef0","resolution":{"observed_at":"2026-08-05T20:10:23.547138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.534979Z","title":"Side adapter network for open-vocabulary semantic segmentation,","venue":null,"work_id":"858bc832-fc9f-4c4e-aab3-4dd0ce9e70a1","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:19.456741Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:7c79cbb8d25fe5a5d5654b98452b150ab75a4408a85c8e8de146ad4a658d6971","observation_id":"b58ed32b-0a2a-4836-886a-5f7a700fb054","resolution":{"observed_at":"2026-08-05T20:10:23.538678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.526487Z","title":"Open-vocabulary panoptic segmentation with text-to-image diffusion models,","venue":null,"work_id":"7e43c89b-998d-42db-bac4-cc9ee1f9a988","year":2023},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:19.565524Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:702fedc334041f154ae9298252abdbecaad086ed5dfef5ffc153feeec7ac5ba9","observation_id":"b08179dd-ad6b-48fa-bdd1-4767272c1b8f","resolution":{"observed_at":"2026-08-05T20:10:23.528893Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.517216Z","title":"Convolutions die hard: Open-vocabulary segmentation with single frozen convolutional clip,","venue":null,"work_id":"853a0c2d-64b9-4b69-9a44-e1657cba1b37","year":2024},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:19.641618Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:eafeabbe798ff714259b3ca09dee4b2307859f94bb8b3c5460d9a92d98483d1f","observation_id":"d6babb57-d5f7-4925-be5f-4a8e1d18117f","resolution":{"observed_at":"2026-08-05T20:10:23.520963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:10:23.508745Z","title":"Extract free dense labels from clip,","venue":null,"work_id":"f69b51af-3041-43f4-b20d-38792ba0b640","year":2022},"citing_paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception","version":1},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-08-05T20:10:19.724032Z"},"links":{"citing_paper":"/paper/2508.11256"},"observation_digest":"sha256:aff073a7ddf5c6814947f5741c7671d295944dcd1ef1af0b4e0384c9ea83ecb1","observation_id":"7482feaa-5a78-4153-a5c2-3bf695e44b61","resolution":{"observed_at":"2026-08-05T20:10:23.511695Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.11256","last_updated":"2025-08-15T06:43:51Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T05:04:02.338549Z","submitted_at":"2025-08-15T06:43:51Z","title":"Generalized Decoupled Learning for Enhancing Open-Vocabulary Dense Perception"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":36,"verified_exact":0,"verified_fuzzy":64},"total_outbound_references":116},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 100 of 116 outbound references and 4 inbound Pith citation observations for arXiv:2508.11256."}