{"as_of":"2026-08-10T04:58:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:15a662a117e43a621f99d398a0645291914955cfb0575db2cb13d1f48783025d","coverage":[{"denominator":57,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":57,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:59:44.658585Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.09612/citation-record","integrity":"/paper/2507.09612/integrity","json":"/paper/2507.09612/citation-record.json","paper":"/paper/2507.09612"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:52.030116Z","title":"Ef- ficient interactive annotation of segmentation datasets with polygon-rnn++","venue":null,"work_id":"bc783c60-78a7-4cfb-abc4-f71f8ec1801a","year":2018},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.050124Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:0b7a8062335949e36b561c07613e34a7d475ddd778a4fae81097a5e126366b08","observation_id":"68897891-50d7-4487-9c27-94c9d5301552","resolution":{"observed_at":"2026-08-06T17:59:52.095454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.904729Z","title":"Moinul Hos- sain, Gianluca Marcelli, Marc Alemany-Fornes, and Anas- tasios D","venue":null,"work_id":"17966069-5d9c-4b9b-b846-cd44c86b753f","year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.169344Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:8eb1674725a9f4ac0c32320ab7b071d1b68d995022cf9ca7e667cb227195a1da","observation_id":"2f1202f6-7312-49d3-af09-fceca1a60c77","resolution":{"observed_at":"2026-08-06T17:59:51.968710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.772405Z","title":"Mvtec ad–a comprehensive real-world dataset for unsupervised anomaly detection","venue":null,"work_id":"35ea596d-d8c7-4d1b-a887-af2b0c353186","year":2019},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.222264Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:73c8566bb54a8d524a092c7bac6a2238f7e3a9f1d5e2bfd2e7c15463d7e4796e","observation_id":"c6f49b15-f40c-4890-965d-70039fad130e","resolution":{"observed_at":"2026-08-06T17:59:51.811180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:40.301798Z","title":"nuscenes: A multi- modal dataset for autonomous driving","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.301798Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:866f7abb90e15e84dc1ce1311cf0d5acc8f2eb8d1852d4ab48d8c6cfdc20eb4e","observation_id":"f3e79b69-9c14-49c2-8301-6d8071ed9aef","resolution":{"observed_at":"2026-08-06T17:59:40.301798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.612281Z","title":"Focalclick: towards practical in- teractive image segmentation","venue":null,"work_id":"70b5c949-4c22-4fc3-8315-79cea1ef6444","year":2022},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.359721Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:839a1f0e71770fd7be227edb6418692361bb115bdef6c01b88c211b8f1fb5d5b","observation_id":"422fcf44-c7d9-4422-abe7-aea092f3e420","resolution":{"observed_at":"2026-08-06T17:59:51.664329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.474926Z","title":"FlashAttention-2: Faster attention with better par- allelism and work partitioning","venue":null,"work_id":"6f35eb9e-339a-4495-8da0-ecf87cbfa548","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.419430Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:63890f2dd0287b98a415de8932675e1c9c4587d17cf0d43f8d7448c9b33679cc","observation_id":"c6486f07-cae1-4c13-832c-4ca56f0c00c7","resolution":{"observed_at":"2026-08-06T17:59:51.540547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.285302Z","title":"Fu, Stefano Ermon, Atri Rudra, and Christopher R´e","venue":null,"work_id":"8741f232-ca7a-4bc5-9760-11ad6f800818","year":2022},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.535287Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:90cf996a8e99ec521835dfd8c2f67cdbb351086276297c2d7df611d85558d316","observation_id":"1b607509-86ef-4e08-9216-2426db8d0f45","resolution":{"observed_at":"2026-08-06T17:59:51.368068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:40.649265Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.649265Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:594a4d7615d64e735e45c92dbb0a3c3da5ae3eb61ad4577c2d5d4ad81e5e3644","observation_id":"4dd43549-5bfc-4f2c-b6f5-c0203e2373ed","resolution":{"observed_at":"2026-08-06T17:59:40.649265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.07372","last_updated":"2023-11-06T13:16:00Z","snapshot_observed_at":"2026-08-09T23:19:57.473713Z","submitted_at":"2022-12-14T17:50:39Z","title":"Image Compression with Product Quantized Masked Image Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.07372","snapshot_observed_at":"2026-08-06T17:59:40.830763Z","title":"Image com- pression with product quantized masked image modeling","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.830763Z"},"links":{"cited_paper":"/paper/2212.07372","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:75b8fc5a8053b69e7d539d69182ac07c7b151df6f482c6a8107633b65c269e30","observation_id":"949d9b8e-f720-4f8a-bd42-6d604bd9ce91","resolution":{"observed_at":"2026-08-06T17:59:40.830763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:40.948848Z","title":"Taming transformers for high-resolution image synthesis","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.948848Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:045f2cafeb751bd904365dff481e040c6f2b91ffe0dc47e9a4b96a6432aebe50","observation_id":"2a939be7-30af-4e14-bb95-8efc9d34d861","resolution":{"observed_at":"2026-08-06T17:59:40.948848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:41.025078Z","title":"Eva: Exploring the limits of masked visual representa- tion learning at scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.025078Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:b15c8e4671d792fec60116343a48131237cb280f2262a3f96bb7a08035400154","observation_id":"9cd767cd-5111-4302-83f8-a573f826ccab","resolution":{"observed_at":"2026-08-06T17:59:41.025078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00752","last_updated":"2024-05-31T17:55:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-01T18:01:34Z","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00752","snapshot_observed_at":"2026-08-06T17:59:41.141507Z","title":"Mamba: Linear-time sequence modeling with selective state spaces","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.141507Z"},"links":{"cited_paper":"/paper/2312.00752","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:fed64eefef62098afa1a1cde2e8cc135339cebd6de0a891b70ca292d02b6d364","observation_id":"faa02641-e8c6-4512-b4f3-9c82e76349ab","resolution":{"observed_at":"2026-08-06T17:59:41.141507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.880213Z","title":"Star-transformer","venue":null,"work_id":"e9de46e6-2f57-4240-a037-f6217fae3267","year":2019},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.197247Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:574560bf49b5e79519f54d5a00f82574579f06579ea21da84101f85269482c17","observation_id":"57937788-8758-4a72-a04e-e97bf210e2da","resolution":{"observed_at":"2026-08-06T17:59:50.946490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.708601Z","title":"Girshick","venue":null,"work_id":"447c3316-ace0-40a6-9004-c227172e593d","year":2019},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.326298Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:e51261875990f9ee94bdebec56964f7b8a8f7a2f23585adcb5cdd4ad39a481a6","observation_id":"6bc10f52-6175-4371-a10a-c8ffc204c68f","resolution":{"observed_at":"2026-08-06T17:59:50.779479Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.554125Z","title":"Flatten transformer: Vision transformer using focused linear attention","venue":null,"work_id":"de008ce0-95ba-4f5b-ac2a-fa405ae325fd","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.414485Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:4282512c18c9c7a138306663c31ad0a1ca66396f90b8f049a26eac9450cf6bb9","observation_id":"23044086-259f-4648-97ec-2fe855688be0","resolution":{"observed_at":"2026-08-06T17:59:50.601022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:41.495681Z","title":"Deep residual learning for image recognition","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.495681Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:e9f48fd872cbce9201c56156e24d1a0d93295b679f3e6223d36faadb3c2edf06","observation_id":"0c8ae67b-9f9f-4a9a-943c-2feadba750c2","resolution":{"observed_at":"2026-08-06T17:59:41.495681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.365951Z","title":"Mask r-cnn","venue":null,"work_id":"5c188f96-25b0-41d7-bf14-3c6f5d40003e","year":2017},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.618278Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:59d8d0fea6844f334adfb7d853e19247f87dd6d1bdafa4802965ce091f7824e6","observation_id":"f48e18d7-fa88-41be-be1d-a6504c5142d2","resolution":{"observed_at":"2026-08-06T17:59:50.460850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.212780Z","title":"Interformer: Real-time interactive image segmentation","venue":null,"work_id":"3db5db1c-8257-4a4b-981f-bdcc27f15ee7","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.742838Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:894b9ecd47b2d9b60becf1bf49ce3429751cddcd4a0317904949b1c7e3c5aa35","observation_id":"18b83410-5428-4233-b66d-4c4ca993df9e","resolution":{"observed_at":"2026-08-06T17:59:50.293455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.02109","last_updated":"2024-11-23T01:44:00Z","snapshot_observed_at":"2026-08-10T03:58:28.648215Z","submitted_at":"2024-07-02T09:51:56Z","title":"HRSAM: Efficient Interactive Segmentation in High-Resolution Images","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.02109","snapshot_observed_at":"2026-08-06T17:59:41.859791Z","title":"Hrsam: Efficiently seg- ment anything in high-resolution images","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.859791Z"},"links":{"cited_paper":"/paper/2407.02109","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:9e1430bd068eb5be177f6076edbffc90be247c8927972df4b0a13d9f7d38a8e2","observation_id":"5e659fb3-e108-49b7-bc20-9d25b4b55524","resolution":{"observed_at":"2026-08-06T17:59:41.859791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.052272Z","title":"Interactive image seg- mentation via backpropagating refinement scheme","venue":null,"work_id":"8c38c54d-3da9-4e30-a6c6-352f980a138e","year":2019},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.979461Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:c5b17bc366cea126b9fd334b92a6bf4bfdbdecd0125c68768b2704781d22b4be","observation_id":"039b08ca-caff-40c4-b4f9-45d403c82f10","resolution":{"observed_at":"2026-08-06T17:59:50.105380Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:49.855488Z","title":"Segment anything in high quality","venue":null,"work_id":"92972ca8-a965-4649-9161-48f95c3a8055","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.093943Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:e13a61f3c07c7f1f86a05f5f5ca9001c68e20364069f30d67ab831c4efffd8a6","observation_id":"13a2bf29-6b91-4c73-b32b-0be3d0d4d076","resolution":{"observed_at":"2026-08-06T17:59:49.948278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:49.702041Z","title":"Transformers in vision: A survey","venue":null,"work_id":"68f527c9-3bdb-464d-8939-2a06eb06590c","year":2022},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.160848Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:781b07e6088d43a317f4bfe029883907ec34caa0a05bead98d434c57530cc0ff","observation_id":"721247d7-4a60-41e1-9626-66566a199402","resolution":{"observed_at":"2026-08-06T17:59:49.779183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:49.516526Z","title":"Segment any- thing","venue":null,"work_id":"14fc30ff-5a71-43db-93ff-4130f0e2c564","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.279968Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:94e9244179e8516df3e4bf0d366512804516e4ffc82452d5c5dc14ea7f9eb301","observation_id":"510cd5c1-3fb1-4a8b-a843-6cc3ccbcebbc","resolution":{"observed_at":"2026-08-06T17:59:49.583175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:49.196204Z","title":null,"venue":null,"work_id":"81685265-6d19-4b8c-98db-f4113bbf14da","year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.391399Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:f01d4461e1a4e841dd629b79b19d6086ed9748fa13b90b28d0edcc654585adc3","observation_id":"e2bdc49e-006c-4078-8871-b0af4fae5d64","resolution":{"observed_at":"2026-08-06T17:59:49.360698Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.00692","last_updated":"2024-05-01T05:10:13Z","snapshot_observed_at":"2026-08-06T07:43:56.889679Z","submitted_at":"2023-08-01T17:50:17Z","title":"LISA: Reasoning Segmentation via Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.00692","snapshot_observed_at":"2026-08-06T17:59:42.435568Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.435568Z"},"links":{"cited_paper":"/paper/2308.00692","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:c9b75a1b0f92d750dec8f8627989ecb6cc73310d14fedcd3a8be029c3c0de770","observation_id":"e1982561-f511-4d16-a33c-2e135b68235f","resolution":{"observed_at":"2026-08-06T17:59:42.435568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:48.738371Z","title":"Exploring plain vision transformer backbones for object de- tection","venue":null,"work_id":"c4d01c60-c2f0-4955-9d9c-c06bd052d024","year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.488811Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:846903c8e3512ee17213f4e13434aa40f23e6eaadbe11031f706173ebf07c979","observation_id":"3ea5ba65-c0c8-4629-ac84-46abbcd06e0b","resolution":{"observed_at":"2026-08-06T17:59:48.980421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:48.576966Z","title":"Interactive image segmentation with latent diversity","venue":null,"work_id":"2062dccb-a689-41b5-945c-7c6f3ebc3988","year":2018},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.557505Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:9ca367c7f946ad2dba488d5fb4a465b835effdce073ba07397f7d513efc0f7ac","observation_id":"fb3b5c69-3955-4d6a-9e84-1d6c548bab83","resolution":{"observed_at":"2026-08-06T17:59:48.618076Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:48.202618Z","title":"Lawrence Zitnick","venue":null,"work_id":"b65ae09d-15e5-4bc6-909d-156401df45e3","year":2014},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.727012Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:191f3c5ace0ad5cdaf1a9e6ec8b9449e9bc59f96888f40ebb41fa1dcb87bcc35","observation_id":"63f3bcea-9b9b-4829-b9ff-37fc53304fd3","resolution":{"observed_at":"2026-08-06T17:59:48.323770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:47.925432Z","title":"Interactive image segmentation with first click attention","venue":null,"work_id":"759d075e-255f-48be-b4ee-c20194009a77","year":2020},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.789996Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:248aa9ea5736fb430b0b784d665b5112ec0cc4dd2cf7548b5bf1b932cf1b63cb","observation_id":"c452a0a2-fb5c-4b5d-b66f-39c214892251","resolution":{"observed_at":"2026-08-06T17:59:48.043298Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16354","last_updated":"2024-02-25T09:48:53Z","snapshot_observed_at":"2026-07-06T16:24:53.814969Z","submitted_at":"2023-09-28T11:26:52Z","title":"Transformer-VQ: Linear-Time Transformers via Vector Quantization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16354","snapshot_observed_at":"2026-08-06T17:59:42.865077Z","title":"Transformer-vq: Linear-time transformers via vector quantization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.865077Z"},"links":{"cited_paper":"/paper/2309.16354","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:04c2829b4fdf14aa33e72ce157e9bd70ea9aa6c9d2ccf7f310e796dc27a60124","observation_id":"ca0054aa-37ef-4ed8-aa3c-6908b3dafa19","resolution":{"observed_at":"2026-08-06T17:59:42.865077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T17:59:42.964375Z","title":"Deepseek-v3 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.964375Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:3ea48509f59c219f0cf1535218d1243f0cff9c3dc97d8146a67942b40ca5900a","observation_id":"329fe068-ed4e-49ae-989b-135a43cbe3a6","resolution":{"observed_at":"2026-08-06T17:59:42.964375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.11006","last_updated":"2023-03-11T19:36:34Z","snapshot_observed_at":"2026-08-09T01:46:23.931777Z","submitted_at":"2022-10-20T04:20:48Z","title":"SimpleClick: Interactive Image Segmentation with Simple Vision Transformers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.11006","snapshot_observed_at":"2026-08-06T17:59:43.065723Z","title":"Simpleclick: Interactive image segmentation with sim- ple vision transformers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.065723Z"},"links":{"cited_paper":"/paper/2210.11006","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:23e81ab6cd3f332294f5efa340662d1998b25689298c572f096c6d78735f12dc","observation_id":"71ee2ea2-52b4-4cea-ba5f-8a4c8bd46526","resolution":{"observed_at":"2026-08-06T17:59:43.065723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:47.684181Z","title":"Pseudoclick: Interactive image segmentation with click imi- tation","venue":null,"work_id":"3872b20a-1326-49e9-a065-d9620f0a6e5b","year":2022},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.122967Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:c9492615d8bbed847a494676476f037a08e2afc4b0d397b00f7c8d966f9f2661","observation_id":"0a4c9921-87d5-4a7a-b679-54aca9b73b6a","resolution":{"observed_at":"2026-08-06T17:59:47.805495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.00741","last_updated":"2024-03-31T17:02:24Z","snapshot_observed_at":"2026-07-06T17:53:44.044640Z","submitted_at":"2024-03-31T17:02:24Z","title":"Rethinking Interactive Image Segmentation with Low Latency, High Quality, and Diverse Prompts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.00741","snapshot_observed_at":"2026-08-06T17:59:43.181168Z","title":"Rethinking interactive image segmentation with low latency, high quality, and diverse prompts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.181168Z"},"links":{"cited_paper":"/paper/2404.00741","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:3b8ad2aec4870a666ddf7d61a248d6f45ba929986b1fd193c9bb61851fd1bb0a","observation_id":"ffa45373-8032-4bee-922d-e432b9bc3a64","resolution":{"observed_at":"2026-08-06T17:59:43.181168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:43.255738Z","title":"Swin transformer: Hierarchical vision transformer using shifted windows","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.255738Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:b7f1f78e5a24c704764fba8c61b00a59ee5cb9a13bc7553eb05417720286283e","observation_id":"b93184ae-072a-40f8-9aca-da083c05f840","resolution":{"observed_at":"2026-08-06T17:59:43.255738Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.12306","last_updated":"2024-04-01T16:18:16Z","snapshot_observed_at":"2026-07-06T15:19:28.033750Z","submitted_at":"2023-04-24T17:56:12Z","title":"Segment Anything in Medical Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.12306","snapshot_observed_at":"2026-08-06T17:59:43.355075Z","title":"Segment anything in medical images","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.355075Z"},"links":{"cited_paper":"/paper/2304.12306","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:f94c60d35b71367cf221e6dfb2182e127ada6a845f2cad3d2a3cffdbecbf31bf","observation_id":"7ae41f4b-92fe-4efb-b2fe-640468955fb8","resolution":{"observed_at":"2026-08-06T17:59:43.355075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:47.341509Z","title":"Deep extreme cut: From extreme points to object segmentation","venue":null,"work_id":"c5b2a1db-6141-48de-8d8c-4735ee280cd5","year":2018},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.451734Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:061ab5c49ff62bbcb5cf10615d186f4dfb55a6ccb70011d06164481a94817bf6","observation_id":"73251273-ef52-4c90-b8e4-ba5c479e7bf0","resolution":{"observed_at":"2026-08-06T17:59:47.500614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:43.527111Z","title":"Segment anything model for medical image analysis: an experimental study","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.527111Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:ec113848216f9063653f9216e40be56349e15168851620ca09fbea431b691317","observation_id":"0443b593-47dc-4ad8-8ce5-8aae21380599","resolution":{"observed_at":"2026-08-06T17:59:43.527111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.15505","last_updated":"2023-10-12T07:55:05Z","snapshot_observed_at":"2026-07-06T16:24:17.829828Z","submitted_at":"2023-09-27T09:13:40Z","title":"Finite Scalar Quantization: VQ-VAE Made Simple","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.15505","snapshot_observed_at":"2026-08-06T17:59:43.608657Z","title":"Finite scalar quantization: Vq-vae made simple","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.608657Z"},"links":{"cited_paper":"/paper/2309.15505","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:fc54558005fa3f05d0f73960ad3d5434e4a2b5c4a216fa420f6ebcc2b92932f0","observation_id":"b035f61b-7131-4304-81e5-9159d52efd71","resolution":{"observed_at":"2026-08-06T17:59:43.608657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:47.055106Z","title":"Gross, and Alexander Sorkine- Hornung","venue":null,"work_id":"ae898552-9830-4e24-8cdf-f00e33b20f62","year":2016},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.681941Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:eec218396dbc8574645cf60366f7c6f2150b9e8b07942181ae6c1f658239fa25","observation_id":"8ed33318-551b-4119-b464-9482ebe37a36","resolution":{"observed_at":"2026-08-06T17:59:47.187738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.05682","last_updated":"2022-10-10T16:36:47Z","snapshot_observed_at":"2026-08-06T17:49:01.053607Z","submitted_at":"2021-12-10T17:25:07Z","title":"Self-attention Does Not Need $O(n^2)$ Memory","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.05682","snapshot_observed_at":"2026-08-06T17:59:43.755282Z","title":"Rabe and Charles Staats","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.755282Z"},"links":{"cited_paper":"/paper/2112.05682","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:037c0d5127b965aa448dd9962286ff257b832090c0c11d6b11d9cf53733d01b6","observation_id":"38941a85-0e27-44a0-bfcd-2a070ef9bc96","resolution":{"observed_at":"2026-08-06T17:59:43.755282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:46.688494Z","title":"Petrov, Olga Barinova, and Anton Konushin","venue":null,"work_id":"16168a43-5f7b-4e0b-80dd-2c9033e50460","year":2020},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.827520Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:50596ba7c8a5dec6aca4cb18995321fa360ac7f9ba27184d5eb371f1daaa1407","observation_id":"a0001edd-c4e3-402b-aed2-3735caf3e73c","resolution":{"observed_at":"2026-08-06T17:59:46.839249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:43.915454Z","title":"Roformer: Enhanced transformer with rotary position embedding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.915454Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:69d3690e594a359578f77bf0c28c2cb546da206194b6f29863254ea79bc4bb4b","observation_id":"6fc9de30-9d6a-4270-b772-4fe39c8996b9","resolution":{"observed_at":"2026-08-06T17:59:43.915454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:43.974606Z","title":"Neural discrete representation learning","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.974606Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:3f75db6dce116b2e8581111400e84a5169117e5379918417ed154d581966fd8b","observation_id":"97a92bc1-5798-4a02-95f0-859d027b5316","resolution":{"observed_at":"2026-08-06T17:59:43.974606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:46.377093Z","title":"Gomez, Lukasz Kaiser, and Illia Polosukhin","venue":null,"work_id":"4366fea4-8dd9-416f-9e56-33e7c9b53575","year":2017},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.042802Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:2df71aea114764bb65912c51d7ef9ecf18a0349ca74a53086e318144eeeb4283","observation_id":"7adda1d5-3e5b-4f11-be44-cc5cc766deca","resolution":{"observed_at":"2026-08-06T17:59:46.517556Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12214","last_updated":"2025-02-06T22:16:59Z","snapshot_observed_at":"2026-08-06T15:36:55.343925Z","submitted_at":"2024-10-16T04:19:28Z","title":"Order-aware Interactive Segmentation","version":3},"cited_work":{"arxiv_id":"2410.12214","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.12214","snapshot_observed_at":"2026-08-06T17:59:45.128674Z","title":"Order-aware Interactive Segmentation","venue":"cs.CV","work_id":"b5a378bf-8c7b-4183-ab5c-cb1c28927e22","year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.103291Z"},"links":{"cited_paper":"/paper/2410.12214","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:9ac5c27c69f9c0a473bfa465d49531b747ed0870489afe7a2ecb3cf0c2b227e0","observation_id":"e37945da-c806-43af-9c51-5b8ecaafa336","resolution":{"observed_at":"2026-08-06T17:59:45.174724Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:46.061879Z","title":"Internimage: Exploring large-scale vi- sion foundation models with deformable convolutions","venue":null,"work_id":"39c5fc08-e6c4-41eb-a646-174eb100db9b","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.162339Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:bb101bba65df69037c06ca75962e1ced566fbd0b9409545e03eb6d9dc451b3e7","observation_id":"b83b033a-e8a4-46d1-85a2-ecaef160dbff","resolution":{"observed_at":"2026-08-06T17:59:46.188030Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00863","last_updated":"2023-12-01T18:31:00Z","snapshot_observed_at":"2026-07-06T16:55:49.813116Z","submitted_at":"2023-12-01T18:31:00Z","title":"EfficientSAM: Leveraged Masked Image Pretraining for Efficient Segment Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00863","snapshot_observed_at":"2026-08-06T17:59:44.221558Z","title":"Efficientsam: Leveraged masked image pretraining for efficient segment anything","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.221558Z"},"links":{"cited_paper":"/paper/2312.00863","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:7cf12fd2519f95022c358b6e746cb17b13795754d64b6024fb4cb057a78623e5","observation_id":"ee6d4960-afe5-4b1c-9271-2392a7a93bad","resolution":{"observed_at":"2026-08-06T17:59:44.221558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04009","last_updated":"2024-05-07T04:57:25Z","snapshot_observed_at":"2026-08-06T00:06:25.546349Z","submitted_at":"2024-05-07T04:57:25Z","title":"Structured Click Control in Transformer-based Interactive Segmentation","version":1},"cited_work":{"arxiv_id":"2405.04009","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.04009","snapshot_observed_at":"2026-08-06T17:59:44.942607Z","title":"Structured Click Control in Transformer-based Interactive Segmentation","venue":"cs.CV","work_id":"a6809a26-d379-4ae3-b5c8-8cb5af31d486","year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.282657Z"},"links":{"cited_paper":"/paper/2405.04009","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:8ddfa23f709a941b52e8fcc34960a69c073afeddb8e9e405d7aa1733cf03601b","observation_id":"db734360-93ec-4cb0-be7c-86b44932a057","resolution":{"observed_at":"2026-08-06T17:59:45.006158Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:45.767750Z","title":"Price, Scott Cohen, Jimei Yang, and Thomas S","venue":null,"work_id":"e18dfc9c-f744-4ef6-a79b-2d8be1e12e05","year":2016},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.339765Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:ee676ea6a92ac43331925c412676b76446840bac0a8c3ed65d234fc2d7356843","observation_id":"c73dd96e-6bd5-4651-bb50-1ddb747eb924","resolution":{"observed_at":"2026-08-06T17:59:45.876912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.04627","last_updated":"2022-06-05T01:57:58Z","snapshot_observed_at":"2026-07-06T11:56:10.689708Z","submitted_at":"2021-10-09T18:36:00Z","title":"Vector-quantized Image Modeling with Improved VQGAN","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.04627","snapshot_observed_at":"2026-08-06T17:59:44.400662Z","title":"Vector-quantized image modeling with improved vqgan","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.400662Z"},"links":{"cited_paper":"/paper/2110.04627","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:9107e543d0dc3345563e5255a5fed6281ad52bbd6c2e74b1d3b8174cf49a6e7c","observation_id":"7ff96d30-366d-477c-b2cf-5c28cf7f9484","resolution":{"observed_at":"2026-08-06T17:59:44.400662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14289","last_updated":"2023-07-01T07:26:22Z","snapshot_observed_at":"2026-08-08T16:17:33.420400Z","submitted_at":"2023-06-25T16:37:25Z","title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14289","snapshot_observed_at":"2026-08-06T17:59:44.470606Z","title":"Faster segment anything: Towards lightweight sam for mo- bile applications","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.470606Z"},"links":{"cited_paper":"/paper/2306.14289","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:5b8483b45b7b88fc9ad1e330cc88e16707cce2550eb7ffd5a340f7363a8b9814","observation_id":"a07228c9-062e-49b5-9d5a-54ab5f4ecd8f","resolution":{"observed_at":"2026-08-06T17:59:44.470606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19423","last_updated":"2024-02-29T18:22:12Z","snapshot_observed_at":"2026-08-03T19:57:28.408015Z","submitted_at":"2024-02-29T18:22:12Z","title":"Leveraging AI Predicted and Expert Revised Annotations in Interactive Segmentation: Continual Tuning or Full Training?","version":1},"cited_work":{"arxiv_id":"2402.19423","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.19423","snapshot_observed_at":"2026-08-06T17:59:44.785740Z","title":"Leveraging AI Predicted and Expert Revised Annotations in Interactive Segmentation: Continual Tuning or Full Training?","venue":"cs.CV","work_id":"3d962c4e-fb25-4df6-b4c4-2def82286471","year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.501671Z"},"links":{"cited_paper":"/paper/2402.19423","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:deb1b55cdf70a26eab7f3258f793d63bcd2f4eaf82eda8ba67a0915b1cf03391","observation_id":"52ab7135-a609-4e82-8ef2-6c08c56889d0","resolution":{"observed_at":"2026-08-06T17:59:44.831422Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07548","last_updated":"2024-06-11T17:59:53Z","snapshot_observed_at":"2026-08-05T21:17:06.014910Z","submitted_at":"2024-06-11T17:59:53Z","title":"Image and Video Tokenization with Binary Spherical Quantization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07548","snapshot_observed_at":"2026-08-06T17:59:44.578141Z","title":"Image and video tokenization with binary spherical quantization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.578141Z"},"links":{"cited_paper":"/paper/2406.07548","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:b20571a366c91fc9206bdf60a22b566982d444c5248f54a8ae06d456b90fc697","observation_id":"6be68bb3-d653-4d4c-832b-eaf9de91d993","resolution":{"observed_at":"2026-08-06T17:59:44.578141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:45.499927Z","title":"Online clustered code- book","venue":null,"work_id":"b477f51e-6651-4d7a-9df4-1ad46ec77426","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.658585Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:a318e2423213cf310ea575120cbb8609eb27e3449f8792a940f58891b2a696ab","observation_id":"25e12b38-6e27-4056-ba6f-6a6f91a887d8","resolution":{"observed_at":"2026-08-06T17:59:45.623460Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:48.420358Z","title":null,"venue":null,"work_id":"b5f595d5-2914-41e1-b3bf-cc819b6d2a7b","year":2018},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":585,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.632091Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:1f33cff912b4111cad184ca28e19a22ff9f6a97f642e9b8cfc56a65be17e60c4","observation_id":"4376c965-97a6-4055-a413-c858a5cfd599","resolution":{"observed_at":"2026-08-06T17:59:48.501036Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.057425Z","title":null,"venue":null,"work_id":"19dce115-8719-447d-8c6b-3472a6f84046","year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.757928Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:bcbb578a0173d06219b9032685b4b28256fc64790842d3d6eeb9c278ca605d9a","observation_id":"886b1b02-97bc-4557-b83c-ed82e9b33a50","resolution":{"observed_at":"2026-08-06T17:59:51.130413Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive"},"reference_resolution":{"displayed":57,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":1,"unresolved":26,"verified_exact":3,"verified_fuzzy":26},"total_outbound_references":57},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 57 of 57 outbound references and 0 inbound Pith citation observations for arXiv:2507.09612."}