{"as_of":"2026-08-10T07:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bb5a18dbcb81a11ca4b6456bdd4db49e90abd17e495e80a22bb2745e48e5306e","coverage":[{"denominator":61,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":61,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:52:49.988237Z","state":"measured"},{"denominator":61,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":61,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.01304/citation-record","integrity":"/paper/2506.01304/integrity","json":"/paper/2506.01304/citation-record.json","paper":"/paper/2506.01304"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.734427Z","title":"Xmem++: Production-level video segmenta- tion from few annotated frames","venue":null,"work_id":"6dab7723-9664-4c39-8532-3bc56666f3b7","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.181841Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:d79090e6fd865d46a732560cd2e5ac6377ac66c38697c885a80fae05eee8d0f8","observation_id":"11a488a3-9d79-45bd-84bf-3beb1059ecc5","resolution":{"observed_at":"2026-08-07T11:52:53.738085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.722117Z","title":"One- shot video object segmentation","venue":null,"work_id":"5829c8e3-d0c2-456d-8a83-b1f07b2497f4","year":2017},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.234531Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:626cf70089790302e13d275ee7c0d35d477c93e1125d96f8d1fc05eef3859df4","observation_id":"366b346b-2fb1-42a3-9b20-82c3210325bf","resolution":{"observed_at":"2026-08-07T11:52:53.726052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.710802Z","title":"Rsprompter: Learning to prompt for remote sensing instance segmenta- tion based on visual foundation model.IEEE Transactions on Geoscience and Remote Sensing, 2024","venue":null,"work_id":"2bb12ff6-8add-4f0f-9e43-55dc09e9a1e7","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.287685Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:27bef274a6114bcba57648146f378f6ce53100604b62a1a21bb27f9b55f9c124","observation_id":"11561c84-e797-4fbf-950d-36de8a62d960","resolution":{"observed_at":"2026-08-07T11:52:53.714060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.699760Z","title":"Adaptformer: Adapting vision transformers for scalable visual recognition","venue":null,"work_id":"d47147b2-0b65-46aa-af1e-5af8dceac751","year":2022},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.368695Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:2c75814ea87bb8ce3fa0f3ba3ac22bac209e9fac5f4fa2c72ad7788ca7205afa","observation_id":"ba02cf08-f5f4-4072-95c3-ddd96b478355","resolution":{"observed_at":"2026-08-07T11:52:53.703563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04579","last_updated":"2024-08-10T11:20:52Z","snapshot_observed_at":"2026-07-06T18:58:30.942582Z","submitted_at":"2024-08-08T16:40:15Z","title":"SAM2-Adapter: Evaluating & Adapting Segment Anything 2 in Downstream Tasks: Camouflage, Shadow, Medical Image Segmentation, and More","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04579","snapshot_observed_at":"2026-08-07T11:52:38.439660Z","title":"Sam2-adapter: Evaluating & adapting segment any- thing 2 in downstream tasks: Camouflage, shadow, medical image segmentation, and more.arXiv:2408.04579, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.439660Z"},"links":{"cited_paper":"/paper/2408.04579","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:061ffe8ae01743e796a5bab26f2401952c0b78d0989d89ff1a085c6bf5ead596","observation_id":"4a334688-76cb-41b5-8674-3a3d106dde9e","resolution":{"observed_at":"2026-08-07T11:52:38.439660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.689355Z","title":"0.1% data makes segment anything slim.NeurIPS,","venue":null,"work_id":"d94cf53c-02f7-4520-8ceb-b29d639f50d1","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.533843Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:73976470d8a381ebfc4e598260d2a199482bb5d0d55f7a5717041d86793bdd5b","observation_id":"db998cdb-5490-40f5-97fe-37adf7115483","resolution":{"observed_at":"2026-08-07T11:52:53.692814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.677987Z","title":"Xmem: Long- term video object segmentation with an atkinson-shiffrin memory model","venue":null,"work_id":"33e59893-66de-40d6-bc58-21dc23460140","year":2022},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.585479Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:c830dc87cc4cf295320fda801a44f66ce91be63ef2a563ca0bf467cfcb0544cd","observation_id":"4c34bce8-2e12-47db-abee-d3958b50182d","resolution":{"observed_at":"2026-08-07T11:52:53.682205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.664897Z","title":"Modular interactive video object segmentation: Interaction-to-mask, propagation and difference-aware fusion","venue":null,"work_id":"b9a4ccf9-d709-48af-bf44-c26902ca759c","year":2021},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.657606Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:1fdec935b727e27506fa0537edc126eedc2d4dc1314f24e216f764e792858e23","observation_id":"4e577a05-e56e-4644-8cdd-8399b2e354d6","resolution":{"observed_at":"2026-08-07T11:52:53.669131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.652440Z","title":"Rethink- ing space-time networks with improved memory coverage for efficient video object segmentation.NeurIPS, 2021","venue":null,"work_id":"14bb0bc5-81b3-4bac-af7b-c1b1c906b0d1","year":2021},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.735290Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:20c041b3a6fe41ec7b158924c84dca3258eb7ef775d4fee6fe61d061390b31a4","observation_id":"7e4ddc88-dcfd-4521-8ca4-3f5f80beee02","resolution":{"observed_at":"2026-08-07T11:52:53.656205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:38.821648Z","title":"Tracking anything with de- coupled video segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.821648Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:05b421c04aa90352815fd0df69d3a17373e3b5d1e19768d8051b7e735ed6b797","observation_id":"f07074bf-67a3-4cd5-8cd0-9247113ec8ae","resolution":{"observed_at":"2026-08-07T11:52:38.821648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.632970Z","title":"Putting the object back into video object segmentation","venue":null,"work_id":"abbc699c-2aae-4361-a6b0-829db8e20020","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.900297Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:a821d85d117415aa5df9d511d4d89845a16749bd00c2475eda55cb977acd33a5","observation_id":"68e2cdb2-f115-42c5-b0f3-2c108187e7a2","resolution":{"observed_at":"2026-08-07T11:52:53.636635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06558","last_updated":"2023-05-11T04:33:08Z","snapshot_observed_at":"2026-08-03T19:49:16.800693Z","submitted_at":"2023-05-11T04:33:08Z","title":"Segment and Track Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06558","snapshot_observed_at":"2026-08-07T11:52:38.984260Z","title":"Segment and track anything.arXiv:2305.06558, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.984260Z"},"links":{"cited_paper":"/paper/2305.06558","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:f50055cfc680884c312b414fbf9df6def385b818ea0dcc0f9aecb773bbbeb776","observation_id":"9b1cc4fd-526c-497a-b21e-a3484ac88c49","resolution":{"observed_at":"2026-08-07T11:52:38.984260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2003.10555","last_updated":"2020-03-23T21:17:42Z","snapshot_observed_at":"2026-08-06T14:37:14.514749Z","submitted_at":"2020-03-23T21:17:42Z","title":"ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2003.10555","snapshot_observed_at":"2026-08-07T11:52:39.082937Z","title":"Electra: Pre-training text encoders as discrimina- tors rather than generators.arXiv:2003.10555, 2020","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.082937Z"},"links":{"cited_paper":"/paper/2003.10555","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:51433e7a82bc18e36e541b23060608154396d8f538a10310452e38dedec30f7c","observation_id":"a33b334d-66b1-4181-a184-740ff8cbc34e","resolution":{"observed_at":"2026-08-07T11:52:39.082937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.622448Z","title":"Learning the what and how of annotation in video object segmentation","venue":null,"work_id":"df732d4d-cdf6-4b71-affe-1285be6e98fc","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.212385Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:37cd4c07f911c4fce64073e5ec2da8aee00ecae2c8e8b3a95a6256a6f5af65a3","observation_id":"409ef4a8-5aeb-4347-a106-dceda66c567b","resolution":{"observed_at":"2026-08-07T11:52:53.625812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00756","last_updated":"2024-08-22T16:38:20Z","snapshot_observed_at":"2026-07-06T18:55:45.493755Z","submitted_at":"2024-08-01T17:57:25Z","title":"Segment anything model 2: an application to 2D and 3D medical images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00756","snapshot_observed_at":"2026-08-07T11:52:39.281260Z","title":"Segment anything model 2: an ap- plication to 2d and 3d medical images.arXiv:2408.00756,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.281260Z"},"links":{"cited_paper":"/paper/2408.00756","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:9e686e01bc60183193155c5c6917b66580a901efbda59447e759bf5f0e967d28","observation_id":"35c29cac-6a12-481e-84a0-ac4ac2ca8add","resolution":{"observed_at":"2026-08-07T11:52:39.281260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-10T01:12:16.468283Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-07T11:52:39.420425Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale.arXiv:2010.11929,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.420425Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:b0de194453639c1da527e7a352883f87e682cd1a87f0f8c0f6a49a7780c20daa","observation_id":"642f041d-7bee-48c2-94c4-6b3bac3f550a","resolution":{"observed_at":"2026-08-07T11:52:39.420425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.611718Z","title":"Interactive video object segmentation using global and local transfer modules","venue":null,"work_id":"aff4b6b9-c607-4d3c-a4cf-985c2f96afef","year":2020},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.603848Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:de04e780d1de7511e5a41d80706dab334ebd0e7295da2f728ccd8fec093ca938","observation_id":"8c099b13-52de-4c6f-820c-5c523858aadf","resolution":{"observed_at":"2026-08-07T11:52:53.615372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.06301","last_updated":"2023-02-17T08:33:28Z","snapshot_observed_at":"2026-08-02T15:18:34.660553Z","submitted_at":"2023-02-13T12:02:51Z","title":"A Neuromorphic Dataset for Object Segmentation in Indoor Cluttered Environment","version":2},"cited_work":{"arxiv_id":"2302.06301","doi":null,"metadata_source":"pith","pith_arxiv_id":"2302.06301","snapshot_observed_at":"2026-08-07T11:52:50.447917Z","title":"A Neuromorphic Dataset for Object Segmentation in Indoor Cluttered Environment","venue":"cs.CV","work_id":"0529405f-902c-4269-ab54-54351c0c87a9","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.861021Z"},"links":{"cited_paper":"/paper/2302.06301","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:f6db4fa6ce4154721fc3c21fc4d28c1d59636a5628e5c54116ee78ec23155e36","observation_id":"58d1537e-41f7-47b0-a316-1aeda0409a70","resolution":{"observed_at":"2026-08-07T11:52:50.520855Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.600557Z","title":"Prompting visual-language models for efficient video understanding","venue":null,"work_id":"b9c30936-f732-44e8-87a5-431324f37815","year":2022},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:40.192127Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:ab50070feb0757b6b07d7adc7049452ddcef06d3de258569662a67e4022236ce","observation_id":"d6944856-87ef-495f-a04c-90eb2ac2d4c0","resolution":{"observed_at":"2026-08-07T11:52:53.604426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.590900Z","title":"Segment anything in high qual- ity.NeurIPS, 2023","venue":null,"work_id":"304cdcf4-d95a-4fb2-bef0-f2510cb29892","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:40.888261Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:cf148447ce9eab9017e598226b7daadcbab239c308d525355a0fd4a85bef0b6d","observation_id":"3c3594b6-e9d5-4f84-86ef-35f0455f93e5","resolution":{"observed_at":"2026-08-07T11:52:53.593770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.581365Z","title":"Segment any- thing","venue":null,"work_id":"8b8d963d-adfe-40de-8f1f-cf7a9695c75c","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:42.580715Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:efcfdea0cec120b560a33e9c9ad9226f2a78cbf4955601724124377fb5e3932f","observation_id":"50853442-a2d1-4ad6-8d6a-4f62cb46ffe0","resolution":{"observed_at":"2026-08-07T11:52:53.584383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.571280Z","title":"Frozen clip models are efficient video learners","venue":null,"work_id":"d28a290f-a247-4a6e-9543-4eb9fc582a58","year":2022},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:44.078694Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:98718185ab3e20727f08227f47f88d968e5d12895a696d61927b9d22b28c48b6","observation_id":"06580f31-8f5e-41a2-a50f-92f1890d0de8","resolution":{"observed_at":"2026-08-07T11:52:53.574762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.07931","last_updated":"2025-03-11T12:57:43Z","snapshot_observed_at":"2026-08-06T05:55:21.187619Z","submitted_at":"2024-08-15T04:59:12Z","title":"Surgical SAM 2: Real-time Segment Anything in Surgical Video by Efficient Frame Pruning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.07931","snapshot_observed_at":"2026-08-07T11:52:44.860444Z","title":"Surgical sam 2: Real-time segment anything in surgical video by efficient frame pruning.arXiv:2408.07931,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:44.860444Z"},"links":{"cited_paper":"/paper/2408.07931","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:63b2725bc58305dc701b512965148a5db61f6f264c35cca877ac643f8c7ec729","observation_id":"3d97282a-0322-42fa-a8d3-8595e413b810","resolution":{"observed_at":"2026-08-07T11:52:44.860444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-09T20:34:52.923500Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-07T11:52:45.975938Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:45.975938Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:8015253faca09f4dd374f5a331d1c06a2dbab6bed1de2b61a3caf59f849d0c59","observation_id":"a27319da-c951-430b-95ce-aa2214fc1cdd","resolution":{"observed_at":"2026-08-07T11:52:45.975938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.553556Z","title":"Segment anything in medical images.Nature Communications, 2024","venue":null,"work_id":"a2981150-d034-45e7-b11b-5151a247219d","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:46.692481Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:77e059b8094f5396ac6cdc5b4b390dc2ac761bface1f0c3bbc00f488f4b8adb8","observation_id":"46538667-b3ba-4a45-9504-1aea96602a29","resolution":{"observed_at":"2026-08-07T11:52:53.562809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.542330Z","title":"Video object segmentation without temporal information.IEEE TPAMI, 2018","venue":null,"work_id":"1d442f76-427f-4c0b-a960-c3d2560c6b49","year":2018},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:46.789541Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:6189f07f6871a9262c927938176ee271337f3be22d03311c4a10c83c30c836e3","observation_id":"296753c5-d4e9-4a3e-9d0c-e1f19ce65fb1","resolution":{"observed_at":"2026-08-07T11:52:53.545539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.463407Z","title":"Segment anything model for medical image analysis: an experimental study","venue":null,"work_id":"7a36c96b-f21d-4a60-88f3-6aabbcebdbae","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:46.896233Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:98a97441cfe5f37d16d6be92dcb9ce9bf855ed5337cbf09588f1ef442c08801f","observation_id":"97ffd79f-007d-4c78-84ff-5e994a6b19fc","resolution":{"observed_at":"2026-08-07T11:52:53.517771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.269816Z","title":"Fast video object segmentation by reference- guided mask propagation","venue":null,"work_id":"ac4ce4cc-d8f0-4f3c-8c73-7a577843630a","year":2018},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.024429Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:0afdb18d41dc343ec17bfe2068cf2edafb22bf3228b226bc4d3b97e3abd57b49","observation_id":"cf47f950-2c8b-4f84-9960-b8b403de068f","resolution":{"observed_at":"2026-08-07T11:52:53.363558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.114877Z","title":"St-adapter: Parameter-efficient image-to-video transfer learning.NeurIPS, 2022","venue":null,"work_id":"e4148757-683f-4c5f-9275-dccd68e25f68","year":2022},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.074668Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:80ca03a57641559e79638029e5bbbba093f31cdfb1b073f0cdffa82ec673320c","observation_id":"ade024f6-0491-46a6-86ce-d70d8e35fd4f","resolution":{"observed_at":"2026-08-07T11:52:53.187973Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.975695Z","title":"Dual- path adaptation from image to video transformers","venue":null,"work_id":"41c7dd3b-6c01-4f12-9a22-ab9e2e0a27c5","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.194040Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:cf0ebb7dc092a313fff2986f3b2cf0254ac0a7742674e1e3158b06db54976323","observation_id":"fce007a0-bd8a-485a-9603-4ae670e231c8","resolution":{"observed_at":"2026-08-07T11:52:53.020964Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:47.273487Z","title":"Pytorch: An imperative style, high-performance deep learning library","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.273487Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:cd88d80899754b4f1d2465aa02e352d924679c289d65d9aedcda77229df22cc3","observation_id":"1c66845a-b417-4fe3-be90-b52433a7d20f","resolution":{"observed_at":"2026-08-07T11:52:47.273487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1704.00675","last_updated":"2018-03-01T17:50:08Z","snapshot_observed_at":"2026-08-02T10:51:13.194643Z","submitted_at":"2017-04-03T16:44:46Z","title":"The 2017 DAVIS Challenge on Video Object Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.00675","snapshot_observed_at":"2026-08-07T11:52:47.338977Z","title":"The 2017 davis challenge on video object segmentation","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.338977Z"},"links":{"cited_paper":"/paper/1704.00675","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:e1c678cbd98c54c152820a7a5d3a5ec1e4652882686e2c14e914dba08c51957b","observation_id":"54246d1f-937a-4143-8f44-ca248c0a50a6","resolution":{"observed_at":"2026-08-07T11:52:47.338977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.935987Z","title":"Disentangling spatial and temporal learning for efficient image-to-video transfer learning","venue":null,"work_id":"2e9ba308-3634-4e9f-8c81-78dd258d543a","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.428687Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:40c11dd32492aae4f2b149345291b732957a9e4bc328aa05462d0c532249ce42","observation_id":"5ba348c6-d502-4d91-869a-c166418a4c9c","resolution":{"observed_at":"2026-08-07T11:52:52.962159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:47.507661Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.507661Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:2acf04413d8b1459b9e3f506b40ca02a9da79afeef16b34349ff30be4d5587bc","observation_id":"4cdc9513-ab3e-4027-a34e-7f0167211ff8","resolution":{"observed_at":"2026-08-07T11:52:47.507661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.01197","last_updated":"2023-12-03T23:57:43Z","snapshot_observed_at":"2026-07-06T15:49:47.586513Z","submitted_at":"2023-07-03T17:58:01Z","title":"Segment Anything Meets Point Tracking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.01197","snapshot_observed_at":"2026-08-07T11:52:47.613125Z","title":"Segment anything meets point tracking.arXiv:2307.01197, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.613125Z"},"links":{"cited_paper":"/paper/2307.01197","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:a780731fc34163986b33db0754420c204c06187ece8940370cc92a4e7d8bb9d6","observation_id":"54567cfc-c79c-4116-b667-2137cf7a270a","resolution":{"observed_at":"2026-08-07T11:52:47.613125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-07T11:52:47.689544Z","title":"Sam 2: Seg- ment anything in images and videos.arXiv:2408.00714,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.689544Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:e7860381dd3d81368c28e4870414bd8a7d9815e4318afe52a4661e5c41504bf1","observation_id":"32d05b03-b271-4e21-ad9f-ba434b00d35e","resolution":{"observed_at":"2026-08-07T11:52:47.689544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.543220Z","title":"Seg- ment anything, from space? InWACV, 2024","venue":null,"work_id":"8734ce5c-9455-4e7a-8fe1-e30d8cbb28ac","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.882759Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:b8007dc6eea1f249f3f1b004e4ae1c1f6503d2f3874095e87f05215eb1311054","observation_id":"b46505a9-ab4f-4412-a50f-b4457b8763be","resolution":{"observed_at":"2026-08-07T11:52:52.634283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.424864Z","title":"Learning fast and robust target models for video object segmentation","venue":null,"work_id":"5f59319b-8202-44fe-81df-0c3b10d91c8b","year":2020},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.977450Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:840084c291b39b424b7594b364c9f77f7d70c391a412a9e90c3af8638cd7024c","observation_id":"d2d55a5b-8207-4979-97ea-d2284c541ef0","resolution":{"observed_at":"2026-08-07T11:52:52.441859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.02635","last_updated":"2025-01-06T03:00:07Z","snapshot_observed_at":"2026-08-10T07:22:11.657664Z","submitted_at":"2024-08-05T16:58:56Z","title":"Interactive 3D Medical Image Segmentation with SAM 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.02635","snapshot_observed_at":"2026-08-07T11:52:48.066282Z","title":"Interactive 3d medical image segmentation with sam 2.arXiv:2408.02635, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.066282Z"},"links":{"cited_paper":"/paper/2408.02635","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:e236c993f7057f0b11f97ef1e8bc8cd120548108fc7497943d64667f64b16bb3","observation_id":"6367c98a-127d-4bb1-8f02-dc5c67f8f2ba","resolution":{"observed_at":"2026-08-07T11:52:48.066282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.13789","last_updated":"2025-01-08T10:01:20Z","snapshot_observed_at":"2026-07-06T17:06:29.162729Z","submitted_at":"2023-12-21T12:26:11Z","title":"TinySAM: Pushing the Envelope for Efficient Segment Anything Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.13789","snapshot_observed_at":"2026-08-07T11:52:48.139202Z","title":"Tinysam: Pushing the envelope for efficient segment any- thing model.arXiv:2312.13789, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.139202Z"},"links":{"cited_paper":"/paper/2312.13789","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:c1117f0931536fb4247a2d31b32668d64ac3605f21de8b5f516ce282d2bb17cb","observation_id":"0af786a1-4ab5-4cc8-94af-fd5a658cad6c","resolution":{"observed_at":"2026-08-07T11:52:48.139202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.352881Z","title":"Towards open-vocabulary video instance segmentation","venue":null,"work_id":"022dbc8c-2559-465a-ac5a-4216eb14fae4","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.219346Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:6d43cc14fb53ba2206e01f9f96f505ba53bd140616cdfb6c04eaff5f8706583d","observation_id":"145770c0-a7d4-449f-ad6e-75da32188b50","resolution":{"observed_at":"2026-08-07T11:52:52.416992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.064899Z","title":"F3net: Fu- sion, feedback and focus for salient object detection","venue":null,"work_id":"c8cb5369-5fc3-4917-82bd-4dc60029365d","year":2020},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.404865Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:6699bd252fa149d9921b310c4e26d2556b54f088bb830d396e84dfe794ff9b2c","observation_id":"a28ac603-12cc-45d1-b433-35d72d35297e","resolution":{"observed_at":"2026-08-07T11:52:52.133490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.12620","last_updated":"2023-12-29T03:40:59Z","snapshot_observed_at":"2026-07-06T15:19:39.223171Z","submitted_at":"2023-04-25T07:34:22Z","title":"Medical SAM Adapter: Adapting Segment Anything Model for Medical Image Segmentation","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.12620","snapshot_observed_at":"2026-08-07T11:52:48.476606Z","title":"Medical sam adapter: Adapting segment anything model for medical image segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.476606Z"},"links":{"cited_paper":"/paper/2304.12620","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:2a1f323e92694ccae052bf6967c9f6895513035ff0607615cc3580e3efa824e1","observation_id":"8fa47d21-827d-415a-8d7b-019097e79d86","resolution":{"observed_at":"2026-08-07T11:52:48.476606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.900798Z","title":"Scalable video object segmentation with simplified frame- work","venue":null,"work_id":"0e0954a2-fb2e-475b-868e-27fadc441cc1","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.565837Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:7b085f8010f8b99a0ab40ea4f24a5a8980977db85c5884b9114d21696be619dd","observation_id":"f66ab37f-aead-4a37-b22e-914904281cd9","resolution":{"observed_at":"2026-08-07T11:52:51.963083Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.879578Z","title":"Cat-sam: Con- ditional tuning for few-shot adaptation of segment anything model","venue":null,"work_id":"c6817636-a9be-471c-ab06-a5b6d0821b02","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.671961Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:20e2b6d1a696bb13d8ab34fa78a4971bd138be2e868f5d4953cb84de57f41979","observation_id":"3cf8432a-bead-4b98-9543-340adc1bda27","resolution":{"observed_at":"2026-08-07T11:52:51.886849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:48.758340Z","title":"Sam2-unet: Segment anything 2 makes strong encoder for natural and medical image segmentation.arXiv:2408.08870,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.758340Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:c9ad921ac9c21c901350d51df6f2f51f79eab1f2899f1bf98e53b32a30eb0c88","observation_id":"2d21e214-7128-4169-94a9-69c759c75ee2","resolution":{"observed_at":"2026-08-07T11:52:48.758340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.862521Z","title":"Efficientsam: Leveraged masked image pretraining for efficient segment anything","venue":null,"work_id":"583c761c-2927-4ab9-abc2-d261561d0350","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.859621Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:16fd67bfe180a6573b02f670298197cdb0a14fb1392be827c2616ccc7a19bd84","observation_id":"f6ec5bef-6694-4bca-b495-7d1b0097f919","resolution":{"observed_at":"2026-08-07T11:52:51.871711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.507461Z","title":"Youtube-vos: Sequence-to-sequence video object segmentation","venue":null,"work_id":"9ad873af-0954-4cb9-be90-a224b49ddb32","year":2018},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.966528Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:7fcdaec0269f5b2c4c49b81abb0535c30ea02a801105b1bedb3aaf9c5672118f","observation_id":"66dd7019-f941-415a-b79a-1d07b6480998","resolution":{"observed_at":"2026-08-07T11:52:51.629829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-10T06:14:13.595413Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-07T11:52:49.056362Z","title":"Track anything: Segment anything meets videos.arXiv:2304.11968, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.056362Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:8e439e6a6e43e58430098771dc9e3151182add24392c5724bfec56506c974b45","observation_id":"9ac3aeba-a833-4a41-8e9e-70ccdfed6daa","resolution":{"observed_at":"2026-08-07T11:52:49.056362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.215096Z","title":"Efficient video object segmen- tation via network modulation","venue":null,"work_id":"0a9564ae-bb80-4f26-b4d0-c9268a7820c7","year":2018},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.155216Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:76cdacc118e36bdd3c085cdcd4a620fd5078bcc7bc84685a5b5583ec72bda9a6","observation_id":"960e79a4-35db-4e07-83ec-aca882a801dc","resolution":{"observed_at":"2026-08-07T11:52:51.325468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.107857Z","title":"AIM: Adapting image models for efficient video action recognition","venue":null,"work_id":"4ea9a138-d8c0-490c-94a6-c9628adb61e4","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.257921Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:efbeb1f54d4da9635ef43a559118608b8941a5ea86da464cbf72f2c0bd3d6154","observation_id":"71beb037-fafc-43c3-8e60-8a94f15c8c43","resolution":{"observed_at":"2026-08-07T11:52:51.156717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:50.961851Z","title":"Collaborative video object segmentation by foreground-background inte- gration","venue":null,"work_id":"467fd5c5-9cd2-4c34-9775-8af0ed92ab05","year":2020},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.351850Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:df17d442ca77270ad207c4518098cb198a73af81cfa6dc0ca5ca25c3b2978827","observation_id":"907785b0-b6af-41a1-98b3-a28ef3db9675","resolution":{"observed_at":"2026-08-07T11:52:51.066669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14289","last_updated":"2023-07-01T07:26:22Z","snapshot_observed_at":"2026-08-08T16:17:33.420400Z","submitted_at":"2023-06-25T16:37:25Z","title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14289","snapshot_observed_at":"2026-08-07T11:52:49.455141Z","title":"Faster segment anything: Towards lightweight sam for mo- bile applications.arXiv:2306.14289, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.455141Z"},"links":{"cited_paper":"/paper/2306.14289","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:7ab02cafa38476e18e9487fba4c50a2a6f8edfba2c3d5e91ab4513e5b95114c9","observation_id":"5404a6f2-5757-4575-aca4-3a46ccf51623","resolution":{"observed_at":"2026-08-07T11:52:49.455141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.09579","last_updated":"2023-12-15T07:21:12Z","snapshot_observed_at":"2026-08-09T19:51:26.770539Z","submitted_at":"2023-12-15T07:21:12Z","title":"MobileSAMv2: Faster Segment Anything to Everything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.09579","snapshot_observed_at":"2026-08-07T11:52:49.547609Z","title":"Mobilesamv2: Faster segment anything to everything.arXiv:2312.09579,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.547609Z"},"links":{"cited_paper":"/paper/2312.09579","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:3d07ef6693e0b5fbff3d738af526fadd09e5a19788e2630f05187cfadcaec63a","observation_id":"b14219f4-0ea2-41f4-8266-4983cc0e55d0","resolution":{"observed_at":"2026-08-07T11:52:49.547609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.13505","last_updated":"2025-04-30T07:19:18Z","snapshot_observed_at":"2026-07-06T16:10:31.087739Z","submitted_at":"2023-08-25T17:30:08Z","title":"Joint Modeling of Feature, Correspondence, and a Compressed Memory for Video Object Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.13505","snapshot_observed_at":"2026-08-07T11:52:49.642417Z","title":"Joint modeling of feature, correspondence, and a compressed memory for video object segmentation.arXiv:2308.13505,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.642417Z"},"links":{"cited_paper":"/paper/2308.13505","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:bd0102b399a0d2a7144f8433b20dba161585e6593346c18faa879aa77cc6a867","observation_id":"2ee8bb8f-c9f7-4a37-b36b-eaaa7dc82e06","resolution":{"observed_at":"2026-08-07T11:52:49.642417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.12156","last_updated":"2023-06-21T10:08:29Z","snapshot_observed_at":"2026-07-06T15:45:01.961896Z","submitted_at":"2023-06-21T10:08:29Z","title":"Fast Segment Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.12156","snapshot_observed_at":"2026-08-07T11:52:49.758396Z","title":"Fast segment any- thing.arXiv:2306.12156, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.758396Z"},"links":{"cited_paper":"/paper/2306.12156","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:5111a002e549dbfc6bebc1467a75e7ac19a937090a07ffeb34c28a51401d1223","observation_id":"a027915a-2f96-4f65-8176-e2d4128ae780","resolution":{"observed_at":"2026-08-07T11:52:49.758396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06660","last_updated":"2025-09-07T19:09:41Z","snapshot_observed_at":"2026-07-06T17:00:00.321552Z","submitted_at":"2023-12-11T18:59:52Z","title":"EdgeSAM: Prompt-In-the-Loop Distillation for SAM","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06660","snapshot_observed_at":"2026-08-07T11:52:49.890149Z","title":"Edgesam: Prompt-in-the-loop distillation for on-device de- ployment of sam.arXiv:2312.06660, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.890149Z"},"links":{"cited_paper":"/paper/2312.06660","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:ac7b771fb0931d1cc68671489e952591e5172597c7022f931df3793f6a18d36a","observation_id":"19b37adb-16b0-4d3f-ad7a-f5ceaa97255e","resolution":{"observed_at":"2026-08-07T11:52:49.890149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00874","last_updated":"2024-12-04T23:51:25Z","snapshot_observed_at":"2026-08-04T09:38:10.661882Z","submitted_at":"2024-08-01T18:49:45Z","title":"Medical SAM 2: Segment medical images as video via Segment Anything Model 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00874","snapshot_observed_at":"2026-08-07T11:52:49.909797Z","title":"Medical sam 2: Seg- ment medical images as video via segment anything model 2.arXiv:2408.00874, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.909797Z"},"links":{"cited_paper":"/paper/2408.00874","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:edc41821d8b4d91afcc594b85794a123026b89df40a99bb5dc11a5d72d197074","observation_id":"d6722ea7-d27d-4d4b-a968-83bc9046d004","resolution":{"observed_at":"2026-08-07T11:52:49.909797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:50.686759Z","title":"-” indicates directly combining existing pre-trained models for inference. “Cost","venue":null,"work_id":"6254e01a-83c2-4054-9032-8d2cb93ca836","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.988237Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:df677b4415a136a9cd630e690f9b435e537633791adddb9aeb763c2051dff5bd","observation_id":"e62b112e-18b2-4ce8-8a43-a0b8f4e6efd4","resolution":{"observed_at":"2026-08-07T11:52:50.813473Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.202306Z","title":null,"venue":null,"work_id":"1251ac23-8615-4c17-bb76-dd20024903f3","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.320982Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:36c1e386b65981aa7619f5ea73c4928ddafa452ee48e8ed06024123424c94455","observation_id":"56a8c35a-5af4-4fc7-8199-6c8eecbe5760","resolution":{"observed_at":"2026-08-07T11:52:52.280421Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.723673Z","title":null,"venue":null,"work_id":"c7b0fd38-1e4a-4537-a1d2-1bff6b80c13f","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.760620Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:495bc12d3816f7203ec64303e96b8ff4a1c1880af00abe9047cc3120ada7fb83","observation_id":"4a59372e-94cc-41ba-9a79-00605792d2b9","resolution":{"observed_at":"2026-08-07T11:52:52.837961Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost"},"reference_resolution":{"displayed":61,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":26,"verified_exact":1,"verified_fuzzy":33},"total_outbound_references":61},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 61 of 61 outbound references and 0 inbound Pith citation observations for arXiv:2506.01304."}