{"as_of":"2026-08-17T18:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:639e7e14892684bb2d5193590551476e3e740b1c8d65e09157f980094b986d5a","coverage":[{"denominator":61,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":61,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T15:47:59.927188Z","state":"measured"},{"denominator":61,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":61,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.13667/citation-record","integrity":"/paper/2501.13667/integrity","json":"/paper/2501.13667/citation-record.json","paper":"/paper/2501.13667"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:01.031022Z","title":"Xmem++: Production-level video segmentation from few annotated frames","venue":null,"work_id":"ab990034-e7aa-4c77-bb36-e8175c4cb0f3","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.638768Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:ae969af7292a66e8322a5e9b1721650f8a40d38629132ba45d41823ef780fdc8","observation_id":"4e1c3f6c-bbc6-4ba5-8b13-9314fc777a8e","resolution":{"observed_at":"2026-08-10T15:48:01.036615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.00263","last_updated":"2020-10-01T09:10:53Z","snapshot_observed_at":"2026-08-16T19:16:38.679239Z","submitted_at":"2020-10-01T09:10:53Z","title":"RefVOS: A Closer Look at Referring Expressions for Video Object Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.00263","snapshot_observed_at":"2026-08-10T15:47:59.644137Z","title":"Refvos: A closer look at referring expressions for video object segmen- tation.arXiv preprint arXiv:2010.00263, 2020","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.644137Z"},"links":{"cited_paper":"/paper/2010.00263","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:e8b60fb32b26a56debb4ff1e52ec5edead6084ac11a77434635228c602a02f2d","observation_id":"bbc84f67-177f-4a4d-9f71-db45046f149d","resolution":{"observed_at":"2026-08-10T15:47:59.644137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:01.013515Z","title":"End-to-end referring video object segmentation with multi- modal transformers","venue":null,"work_id":"b51fa50d-b93e-4244-875f-2fbe6f9e143b","year":2022},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.649698Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:cb74946ef6076f00f7720c84febe8d1be0c4ee258ef6fb0fde7f7f79c270beed","observation_id":"6bf26444-f89a-4366-b158-ec586a8a771a","resolution":{"observed_at":"2026-08-10T15:48:01.019502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.995932Z","title":"End-to-end referring video object segmentation with multi- modal transformers","venue":null,"work_id":"3d4a1a25-89b9-4156-b3af-bc333cd2b8b9","year":2022},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.654829Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:28de9c38dee4723a6051a1c93df66005c6c74564544505b110701bb8d91305fd","observation_id":"fdc8ee81-920a-4426-9562-f90f532b5172","resolution":{"observed_at":"2026-08-10T15:48:01.001502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.978813Z","title":"End- to-end object detection with transformers","venue":null,"work_id":"1c4b89c2-7f12-4897-b9e8-0f621ec60afb","year":2020},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.659696Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:2858c4fc0931ae873213d591aec791d3fe13582bb1215c54d046bc0115a895a4","observation_id":"987ce447-7a12-4674-bc10-519771e68145","resolution":{"observed_at":"2026-08-10T15:48:00.984032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.961807Z","title":"Xmem: Long- term video object segmentation with an atkinson-shiffrin memory model","venue":null,"work_id":"fb027db8-329c-4a61-8a01-b894e77fb9a8","year":2022},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.664751Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:7c2356b1b0c3e841cb118d4e4347cec7f9dadd73c2008f7a2e1d3ad49f5c16e6","observation_id":"755fb4b8-5a47-4de1-9e55-e8b1463f784c","resolution":{"observed_at":"2026-08-10T15:48:00.966930Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.944242Z","title":"Putting the object back into video object segmentation","venue":null,"work_id":"20e6c225-7ab9-48a7-bbf7-a7e550b122a9","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.670130Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:7df83a7eea7bd870ae3a7de41124fc18e6d60081b9ee7bf48d402902f6d91d0f","observation_id":"ff279dd1-d879-4b85-adb0-7c78be7d3270","resolution":{"observed_at":"2026-08-10T15:48:00.950457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06558","last_updated":"2023-05-11T04:33:08Z","snapshot_observed_at":"2026-08-16T15:33:58.868647Z","submitted_at":"2023-05-11T04:33:08Z","title":"Segment and Track Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06558","snapshot_observed_at":"2026-08-10T15:47:59.674573Z","title":"Segment and track anything.arXiv preprint arXiv:2305.06558, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.674573Z"},"links":{"cited_paper":"/paper/2305.06558","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:e1b2181775440eb9d4ece4b771044b837b2422abf97dac85090a72306e4df563","observation_id":"4546e337-6ea6-4a03-8f0c-ec6d3e3a81d0","resolution":{"observed_at":"2026-08-10T15:47:59.674573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1911.02116","last_updated":"2020-04-08T01:02:17Z","snapshot_observed_at":"2026-08-16T12:39:33.749100Z","submitted_at":"2019-11-05T22:42:00Z","title":"Unsupervised Cross-lingual Representation Learning at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.02116","snapshot_observed_at":"2026-08-10T15:47:59.679455Z","title":"Unsupervised cross-lingual representation learning at scale.arXiv preprint arXiv:1911.02116, 2019","venue":null,"work_id":null,"year":1911},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.679455Z"},"links":{"cited_paper":"/paper/1911.02116","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:405c8726b54c2bd16f8a26bae35cde2159a492dc0de51ade263806868bd05b41","observation_id":"055a0632-966b-491b-95e1-cc2a2fbb4493","resolution":{"observed_at":"2026-08-10T15:47:59.679455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.926973Z","title":"Vision-language transformer and query generation for refer- ring segmentation","venue":null,"work_id":"878ff9ab-9cf9-4375-863b-4da1e69ac13a","year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.684899Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:baa43521d1f4f8e5f04b3c7fcd9b365e0491ee3f6016f14edd518bf0193f17cf","observation_id":"ff018f14-f355-42b4-b98e-04a998bfec9a","resolution":{"observed_at":"2026-08-10T15:48:00.932309Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.908046Z","title":"Mevis: A large-scale benchmark for video segmentation with motion expressions","venue":null,"work_id":"8a80f355-90a7-41ea-bb44-13106c6d22b2","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.690233Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:83d8a33c5d6a77acf5ae23bdee2644cc402dae2cbc13101edf75a57066583bec","observation_id":"0fa86d86-2c75-4729-a592-455d0adf2496","resolution":{"observed_at":"2026-08-10T15:48:00.913517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.891826Z","title":"Language-bridged spatial-temporal interaction for referring video object segmentation","venue":null,"work_id":"ec35246f-7850-408a-ad50-511b6e2c84fd","year":2022},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.695176Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:a0ceeb86a39a6e9629d84b2e979deaeaa2b9ee4af457a45d39359351a2836750","observation_id":"5971daea-c441-47f0-b780-85b500c13e4d","resolution":{"observed_at":"2026-08-10T15:48:00.896976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.875855Z","title":"Unified embedding alignment for open-vocabulary video instance segmentation","venue":null,"work_id":"cf992954-c995-423a-ac52-145184ac65df","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.700026Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:34a4569986f4d34ba1ba31b93c1100ef1aea4d2b32be90f7099bb032436a9a5a","observation_id":"f94ac849-ef94-41d6-ab2c-b0ef10281cb4","resolution":{"observed_at":"2026-08-10T15:48:00.881133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.859906Z","title":"Html: Hybrid temporal-scale mul- timodal learning framework for referring video object seg- mentation","venue":null,"work_id":"01449708-d1e9-4a95-810c-f0ed4c0c5067","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.704741Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:5c1e8c2ff861886931c02787b8f7fe6fdd6925ce57175bcf1da9e34dc47145a4","observation_id":"64500324-77b2-49a1-b524-4c5921b9300e","resolution":{"observed_at":"2026-08-10T15:48:00.865317Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.843057Z","title":"Decoupling static and hier- archical motion perception for referring video segmentation","venue":null,"work_id":"a080f141-2f3a-4c27-a393-318d65f509e7","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.709388Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:90bc3364c1e58bc94170c18a600f60704808d80c70bd42c58f55dd1e289a99e7","observation_id":"160ea41c-b019-473c-9647-601d3cdd12c5","resolution":{"observed_at":"2026-08-10T15:48:00.848532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15876","last_updated":"2024-12-23T08:10:30Z","snapshot_observed_at":"2026-08-16T13:23:10.733965Z","submitted_at":"2024-08-28T15:47:32Z","title":"Unleashing the Temporal-Spatial Reasoning Capacity of GPT for Training-Free Audio and Language Referenced Video Object Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15876","snapshot_observed_at":"2026-08-10T15:47:59.714033Z","title":"Unleashing the temporal-spatial reasoning capacity of gpt for training-free audio and language referenced video object segmentation.arXiv preprint arXiv:2408.15876, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.714033Z"},"links":{"cited_paper":"/paper/2408.15876","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:b2cbe770b23ba1a6ba756f95688ed9a1b9816ba26cc02e47d16a9bef13f8763a","observation_id":"7161f8c8-2472-4e51-8763-388d1fb61b66","resolution":{"observed_at":"2026-08-10T15:47:59.714033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.718988Z","title":"Segment anything in high qual- ity.Advances in Neural Information Processing Systems, 36,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.718988Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:245cf00b93be572f89afd3ebe180ab9a9d49213d2e48f0f07bdb0d9b0d1f9606","observation_id":"69c6c784-e421-4ba8-806c-7309be4bb745","resolution":{"observed_at":"2026-08-10T15:47:59.718988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.809057Z","title":"Video object segmentation with language referring expressions","venue":null,"work_id":"af3a1bb4-d5ae-4eda-8524-9f06ae0bfe7d","year":2018},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.723933Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:ed47f3515ea1c788796276237548e49a9386142c214cbdfe71d4424bcc36c748","observation_id":"dfe47324-1f8c-4d7a-aaad-9288711f0ff0","resolution":{"observed_at":"2026-08-10T15:48:00.815303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.792973Z","title":"Segment any- thing","venue":null,"work_id":"7ca258a8-01ed-4ccf-a610-2363e57a1378","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.728435Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:351ef9074dabd2fbce024ab82b87a7aae11a029ecb302bd0dafb07c07cd07ed3","observation_id":"4e3d5cfb-376c-4f68-a950-94455ae29c78","resolution":{"observed_at":"2026-08-10T15:48:00.797935Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.775526Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":"65d16c4d-35d8-452d-9cb1-ed3dd86af795","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.733316Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:12b8ce0737fef8bd59b53d88593fa16ef21760407bc502ab7140677a551ae7c9","observation_id":"b89bf872-4ce4-4000-89c9-2bd0ba6b7222","resolution":{"observed_at":"2026-08-10T15:48:00.780736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.759769Z","title":"Learning to learn better for video object segmentation","venue":null,"work_id":"5327e26d-d0f9-49c5-bd71-612d5993fbd0","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.737900Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:836eabe62cd7f8c61c5f71545690533b8b6272f08b01b02b320d5ab9d71c723a","observation_id":"ced4a7ee-c08d-48ca-af59-a115b9104a63","resolution":{"observed_at":"2026-08-10T15:48:00.764725Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.744169Z","title":null,"venue":null,"work_id":"9790b4f4-881c-4c89-89ba-08f4c8be6d09","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.742330Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:bdbd17c7a2ff049dd227de2228416610fdd3ea1ff02eb00608cce3acfef89184","observation_id":"328add4e-aec4-408f-9b64-772b9545be10","resolution":{"observed_at":"2026-08-10T15:48:00.749283Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.728462Z","title":"Bidirectional correlation-driven inter-frame inter- action transformer for referring video object segmentation","venue":null,"work_id":"35918d2c-a36a-4d7c-ba03-44b12de51c71","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.747025Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:16b37aed81e1e195eea9bb477108669dc40618fcba58ba22f4a6cea93fb94016","observation_id":"5b791fd6-2d71-4e4c-a4a1-6985029cb2a2","resolution":{"observed_at":"2026-08-10T15:48:00.733566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.712885Z","title":"You only infer once: Cross-modal meta-transfer for referring video object segmentation","venue":null,"work_id":"02a7424d-6162-4e41-a379-cbf42e8ecbe4","year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.751536Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:72a6e70e8546e17d80ad1609eb5e97df7a3d95a9f0a1d15705d52d77f9bdd564","observation_id":"9701dfa4-8e8e-4eef-886e-cd73a0ea8d1c","resolution":{"observed_at":"2026-08-10T15:48:00.717969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.00997","last_updated":"2024-09-03T07:25:51Z","snapshot_observed_at":"2026-08-16T15:19:09.668585Z","submitted_at":"2023-07-03T13:21:58Z","title":"RefSAM: Efficiently Adapting Segmenting Anything Model for Referring Video Object Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.00997","snapshot_observed_at":"2026-08-10T15:47:59.756198Z","title":"Refsam: Efficiently adapting segmenting any- thing model for referring video object segmentation.arXiv preprint arXiv:2307.00997, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.756198Z"},"links":{"cited_paper":"/paper/2307.00997","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:dbdfe3b2fc9b45d8494ac0010e8db119a8684069968be0bbee1b93ae19d41d27","observation_id":"e5c6324e-ed1f-44bc-9d23-12fb89929b46","resolution":{"observed_at":"2026-08-10T15:47:59.756198Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.761037Z","title":"Visual instruction tuning.Advances in neural information processing systems, 36, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.761037Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:9925500a0c35ebab0a1456c253006777e114f58afb293ce711594711688000a5","observation_id":"3e9f7077-2c38-4e6e-99f5-159afcd2d098","resolution":{"observed_at":"2026-08-10T15:47:59.761037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-10T15:47:59.765578Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection.arXiv preprint arXiv:2303.05499, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.765578Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:003835c7a8ac0644d0336266404234830dd69f3bbb7e857a0c552ff95dd5995a","observation_id":"fd5bc5ec-19e1-40f6-b456-e6f50d648946","resolution":{"observed_at":"2026-08-10T15:47:59.765578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.770551Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.770551Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:4d0e6a1683f0cb152ddeed5abe374f9859ccc010332afce5896df9d11fb2b718","observation_id":"230a054c-9723-4a4a-93ea-8bf4aac4c325","resolution":{"observed_at":"2026-08-10T15:47:59.770551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.677172Z","title":"Soc: Semantic-assisted object cluster for referring video object segmentation.Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":"de74a9c3-3a11-4ed1-a2c9-80e7e4cc7623","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.775051Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:c4030e161cacc15a7c5cba3d1f6105364494b22fb1b0ff7960f35774ea538732","observation_id":"d947f84c-2012-4c4a-94d4-3e539f1415bd","resolution":{"observed_at":"2026-08-10T15:48:00.682289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.779445Z","title":"Generation and comprehension of unambiguous object descriptions","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.779445Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:b615405253c58a5d1067c8cd5ce47dc1ce381fca1495cec329fce73b63466d82","observation_id":"50a38455-8280-4a0f-9131-321f32e3234f","resolution":{"observed_at":"2026-08-10T15:47:59.779445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.650871Z","title":"Visual-textual capsule routing for text-based video segmentation","venue":null,"work_id":"39a76a48-caac-40ff-b514-6d052f7de5ee","year":2020},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.784047Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:33ed1c29b882c41290d46d791782d1659a5bf614d1b2b83e86130554456992c4","observation_id":"bd1b8c88-a717-449d-a81c-135e74c8e4ab","resolution":{"observed_at":"2026-08-10T15:48:00.655677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.635363Z","title":"Spectrum-guided multi-granularity referring video object segmentation","venue":null,"work_id":"b92db058-377a-4ee7-937f-7fabd02554c9","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.788681Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:ee003e7fd91d458ab426276fed45b3b5b59832005e19ce779d1098612632e754","observation_id":"38e25771-d08d-407a-a24a-19c3b8c25675","resolution":{"observed_at":"2026-08-10T15:48:00.640795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.619907Z","title":"V-net: Fully convolutional neural networks for volumetric medical image segmentation","venue":null,"work_id":"98f17e2f-ea00-4d12-84ca-b2b7ec1db621","year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.793025Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:69c62cc7b7b64ea901a2722ae11201b3073124d4dafac592a20ad31a62d06c1b","observation_id":"04df44cf-137c-43d2-9023-03e4e5a2394f","resolution":{"observed_at":"2026-08-10T15:48:00.624776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.604201Z","title":"Video object segmentation using space-time memory networks","venue":null,"work_id":"97d27f07-7733-4e6c-abc7-43bd4a8e339f","year":2019},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.797817Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:a3346f731a3521798eb01ca502c9128868b8bf755b1960f5917141bda57e9a0a","observation_id":"ad7d8cdc-6449-4416-bddc-15badaeee255","resolution":{"observed_at":"2026-08-10T15:48:00.609152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.588013Z","title":"Semantic and sequential alignment for referring video object segmentation","venue":null,"work_id":"4e4fe5b6-8902-416b-873a-66e18f4b8197","year":2025},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.802324Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:d755b2ba948cb0bcffd7f689568f359b1986421aa1ee9c0734293c551e3aab83","observation_id":"b476880e-13e2-4bc3-ba23-53d3216bf272","resolution":{"observed_at":"2026-08-10T15:48:00.593023Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.00675","last_updated":"2018-03-01T17:50:08Z","snapshot_observed_at":"2026-08-02T10:51:13.194643Z","submitted_at":"2017-04-03T16:44:46Z","title":"The 2017 DAVIS Challenge on Video Object Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.00675","snapshot_observed_at":"2026-08-10T15:47:59.807087Z","title":"The 2017 davis challenge on video object segmentation.arXiv preprint arXiv:1704.00675, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.807087Z"},"links":{"cited_paper":"/paper/1704.00675","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:a3bd8cc1aacb9b07b8ff6d462da552e6dee4cc550b30edb84676ddbec26fc3de","observation_id":"a2d4b4e7-3df5-46c6-88d9-7eb0fb05c1c2","resolution":{"observed_at":"2026-08-10T15:47:59.807087Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.811931Z","title":"Glamm: Pixel grounding large multimodal model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.811931Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:5dd200a51498e65417a97018720413ac3184602bcd4edee78f17e25a49cd69d4","observation_id":"e384604e-d98d-4289-b000-5cd9177d1b1e","resolution":{"observed_at":"2026-08-10T15:47:59.811931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-10T15:47:59.817076Z","title":"Sam 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.817076Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:2679bc78534c328cae47d4a212e168e6b3809963ea33db6942ac0e8d45503e30","observation_id":"36430e9f-d503-4473-b3e4-ed7f044aac35","resolution":{"observed_at":"2026-08-10T15:47:59.817076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.821941Z","title":"Cus- tomized sam 2 for referring remote sensing image segmenta- tion.arXiv preprint arXiv:2503.07266, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.821941Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:3a8d03726a3ca2532046caee1aafb443ac445e660d370432a46c73a639f92958","observation_id":"1dd84681-af88-483f-aac6-521cc1ddbc2e","resolution":{"observed_at":"2026-08-10T15:47:59.821941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.562920Z","title":"Urvos: Unified referring video object segmentation network with a large-scale benchmark","venue":null,"work_id":"4dddb96e-f91d-403d-b63f-f44041a038c7","year":2020},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.826708Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:0198b1dbb6ff277213626aee00136f505c9c5a85f90eda79e1a4711a93912389","observation_id":"92ed7359-2cfb-4b8d-985e-b3e434a333e0","resolution":{"observed_at":"2026-08-10T15:48:00.567693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.547743Z","title":"Temporal collection and distribution for referring video object segmentation","venue":null,"work_id":"9103b88e-1063-40ed-9222-06296f3e6278","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.831198Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:2217709c7d293152554c8407dd76fc53d43559826fe5e2d69b8df40536349926","observation_id":"94e8d278-d03b-4049-8de5-aca91dbf4e0b","resolution":{"observed_at":"2026-08-10T15:48:00.552878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.532861Z","title":"Samrs: Scaling-up re- mote sensing segmentation dataset with segment anything model.Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":"f3f82495-706d-4e25-b873-3c8ab643ac8f","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.835790Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:a385983ea96369b0bd405842c20593792964bb16c19d4c4b01a0ef0ea3e883ce","observation_id":"3065cf5b-7afd-4526-833f-13bb09643fd2","resolution":{"observed_at":"2026-08-10T15:48:00.537906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.515607Z","title":"Asymmetric cross-guided attention network for actor and ac- tion video segmentation from natural language query","venue":null,"work_id":"8965fc92-46d6-4ca6-9dd4-feea4152b7a3","year":2019},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.840270Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:ec6125905b57997d370fd613dc4f3ffed77187e1b5a2f844539296ebc92830fa","observation_id":"a1d6c23c-9577-466b-b38a-327dcc6a4f30","resolution":{"observed_at":"2026-08-10T15:48:00.521970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.499414Z","title":"Image as a foreign language: Beit pretraining for vision and vision- language tasks","venue":null,"work_id":"a07344ed-9ea5-4ba5-8f90-b22c4c899ede","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.845067Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:789537e4e03b9a22cd0362c639f9b436e9df9159e88118d9979377eb494afbaa","observation_id":"5959649d-ce2c-46b4-9e05-e8c3ca604f26","resolution":{"observed_at":"2026-08-10T15:48:00.504760Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17606","last_updated":"2024-12-02T03:19:04Z","snapshot_observed_at":"2026-08-13T21:35:43.932778Z","submitted_at":"2024-11-26T17:18:20Z","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.17606","snapshot_observed_at":"2026-08-10T15:47:59.849646Z","title":"Hyperseg: Towards univer- sal visual segmentation with large language model.arXiv preprint arXiv:2411.17606, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.849646Z"},"links":{"cited_paper":"/paper/2411.17606","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:5803afa24f1f917d8fdda01af43736e6e5097ed1bde93aa6675a6f8335315e10","observation_id":"8f1e15ea-a785-4cc4-ace7-b3b1e43c0986","resolution":{"observed_at":"2026-08-10T15:47:59.849646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.482983Z","title":"Multi-level representation learning with semantic alignment for referring video object segmentation","venue":null,"work_id":"8766dfd0-3656-4dfb-b9ca-61f6d676c214","year":2022},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.854413Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:f0e1e677d6b585c1da1f3a17b1b0707c5dd892bb9f37114130574d2be53f8c9f","observation_id":"99a2b93d-69c9-4020-8819-a18dc653a03a","resolution":{"observed_at":"2026-08-10T15:48:00.488432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.466345Z","title":"Onlinerefer: A simple online baseline for referring video object segmentation","venue":null,"work_id":"9e2bf597-263e-4a99-9a64-7da22d80a0cc","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.858650Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:4ed98ee6c1efcb9788a43457c5b675534972dc124d8e9828bab01872e16d1548","observation_id":"dcaacb23-6094-40d8-a4eb-12d4d8299c9d","resolution":{"observed_at":"2026-08-10T15:48:00.471795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.450579Z","title":"Language as queries for referring video object segmen- tation","venue":null,"work_id":"77c7d988-ddab-4e4d-81e9-ff11b04341af","year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.863235Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:c950dbd8a1d4e10ff21b3b8acac775f5b58a15a8e9760c3f93eee4fe901aa136","observation_id":"27dad8b8-5237-45d9-b191-642f246c9a2d","resolution":{"observed_at":"2026-08-10T15:48:00.455585Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.433653Z","title":"Logiczsl: Exploring logic- induced representation for compositional zero-shot learning","venue":null,"work_id":"aad39284-cb7d-4db5-ad50-e148deda8e5a","year":2025},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.868031Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:3a46f2c3057163db0fc3608ccc43b2e4cd26f1590cdd4ad2c879f30a30573184","observation_id":"af69f5b1-cba9-4ea5-bb47-dfc980093eef","resolution":{"observed_at":"2026-08-10T15:48:00.438701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.417449Z","title":"Efficientsam: Leveraged masked image pretraining for efficient segment anything","venue":null,"work_id":"89bb53a8-4b28-4370-9a79-3ebb2b8e2828","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.873004Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:bfc0826c0ffab6969d956bb88f03130ff37df695b911970423ab401fe97d3365","observation_id":"3690eaf6-802f-4159-9288-f20a2e6b9847","resolution":{"observed_at":"2026-08-10T15:48:00.422489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.05348","last_updated":"2024-08-28T14:26:07Z","snapshot_observed_at":"2026-08-16T14:44:40.421382Z","submitted_at":"2023-11-09T13:18:27Z","title":"u-LLaVA: Unifying Multi-Modal Tasks via Large Language Model","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.05348","snapshot_observed_at":"2026-08-10T15:47:59.877607Z","title":"u-llava: Uni- fying multi-modal tasks via large language model.arXiv preprint arXiv:2311.05348, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.877607Z"},"links":{"cited_paper":"/paper/2311.05348","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:8867a6e970c7b929b0b7814541028defcc6252553581ad319b8504b648b297ad","observation_id":"d52ffb01-c86d-4286-9866-779efd545e76","resolution":{"observed_at":"2026-08-10T15:47:59.877607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.400756Z","title":"Visa: Reasoning video object segmentation via large language models","venue":null,"work_id":"99834d2f-208d-44a8-bbbb-5d0f6430c26e","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.882837Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:a88fe0287ccdf478038bbd7795a60d580291260e327c54ab18af372202063dc7","observation_id":"de3c20d8-e016-42fd-8746-f16c4ab22481","resolution":{"observed_at":"2026-08-10T15:48:00.405801Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.384247Z","title":"Referred by multi-modality: A unified tem- poral transformer for video object segmentation","venue":null,"work_id":"598d03fb-81f0-4730-adc2-e3fa7e16cce5","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.887488Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:3973f12bd6d943985c32a63c391b5421b03618cdc2edff87557b54855fd27763","observation_id":"f36f6979-76b4-4c7c-9717-e82879d29075","resolution":{"observed_at":"2026-08-10T15:48:00.390008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.367370Z","title":"Modeling context in referring expres- sions","venue":null,"work_id":"5b05721c-31c2-4449-9318-2b925f409c88","year":2016},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.892095Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:7584308912278f2a982f02c7d019b14fa834c074f1ac9c2d07c56371cbbe505f","observation_id":"39fab523-0cbf-4cde-a3ea-496bc9ffb894","resolution":{"observed_at":"2026-08-10T15:48:00.372685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15521","last_updated":"2025-06-17T06:34:15Z","snapshot_observed_at":"2026-08-17T09:51:01.392839Z","submitted_at":"2024-08-28T04:14:01Z","title":"A Simple Baseline with Single-encoder for Referring Image Segmentation","version":3},"cited_work":{"arxiv_id":"2408.15521","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.15521","snapshot_observed_at":"2026-08-10T15:47:59.998003Z","title":"A Simple Baseline with Single-encoder for Referring Image Segmentation","venue":"cs.CV","work_id":"53d639f0-c4ba-4522-b0cb-edd8c2ae3901","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.897025Z"},"links":{"cited_paper":"/paper/2408.15521","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:987ddd21b58f069ec8fa6554ae6ba88728dcfb2425d44cc0e0e8f656e91dff14","observation_id":"cc420157-1065-4e3b-a348-dada5e697f3f","resolution":{"observed_at":"2026-08-10T15:48:00.007696Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.901850Z","title":"Losh: Long-short text joint prediction network for referring video object segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.901850Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:c6a75f325978c5cd1ea26d2bdccac98ecbfaa3cd1be641b18c9eed66ef89c149","observation_id":"959aed02-6d41-47d3-9e66-7763d8836a05","resolution":{"observed_at":"2026-08-10T15:47:59.901850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.906540Z","title":"Surgicalsam: Efficient class prompt- able surgical instrument segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.906540Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:d0d908553fd7b941fa1dc35d36c26f8a5fdde8cfc0c7c6a122eeb977920f7679","observation_id":"015c654e-2242-41cf-a9d4-161540a8f6cb","resolution":{"observed_at":"2026-08-10T15:47:59.906540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14289","last_updated":"2023-07-01T07:26:22Z","snapshot_observed_at":"2026-08-12T07:11:05.316372Z","submitted_at":"2023-06-25T16:37:25Z","title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14289","snapshot_observed_at":"2026-08-10T15:47:59.911548Z","title":"Faster segment anything: Towards lightweight sam for mo- bile applications.arXiv preprint arXiv:2306.14289, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.911548Z"},"links":{"cited_paper":"/paper/2306.14289","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:fccefa171754b433f086666f0fc2697a3c0928d8fa8387fd5df7b2b05fbb058a","observation_id":"42fa24c1-7262-43ff-885e-fd50ec4fb259","resolution":{"observed_at":"2026-08-10T15:47:59.911548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.20076","last_updated":"2025-03-10T12:34:24Z","snapshot_observed_at":"2026-08-17T08:49:26.984989Z","submitted_at":"2024-06-28T17:38:18Z","title":"EVF-SAM: Early Vision-Language Fusion for Text-Prompted Segment Anything Model","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.20076","snapshot_observed_at":"2026-08-10T15:47:59.916575Z","title":"Evf- sam: Early vision-language fusion for text-prompted seg- ment anything model.arXiv preprint arXiv:2406.20076,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.916575Z"},"links":{"cited_paper":"/paper/2406.20076","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:170d806cbf910024e743edbff608f080f4d083f51cac063497f9f04db8278dff","observation_id":"747b87e0-fc36-405d-a4e0-52a21ae4e60d","resolution":{"observed_at":"2026-08-10T15:47:59.916575Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.922407Z","title":"Deformable detr: Deformable transformers for end-to-end object detection","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.922407Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:d1ee719355c586e4d6f55d6af066984d3e73030e30b8b2a6e44aca24e9632f20","observation_id":"a569ecdb-9f60-4755-9229-2bedc8ca6b6c","resolution":{"observed_at":"2026-08-10T15:47:59.922407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.320204Z","title":"Exploring pre-trained text- to-video diffusion models for referring video object segmen- tation","venue":null,"work_id":"3b9fc9f1-ac7a-4741-8dd6-8591355af344","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.927188Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:47aaf4d8793ad1f0b30ec671319bd631ac6511850d63d4cdd04c97071bede35f","observation_id":"0cd8a6d0-c9f7-4798-b2df-3d797451fb83","resolution":{"observed_at":"2026-08-10T15:48:00.326524Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","latest_version":5,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-17T05:01:26.818716Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation"},"reference_resolution":{"displayed":61,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":22,"verified_exact":1,"verified_fuzzy":38},"total_outbound_references":61},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 61 of 61 outbound references and 0 inbound Pith citation observations for arXiv:2501.13667."}