{"as_of":"2026-08-10T12:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cef7c42d4f4c85fa644bedbf4b00300c3e5bd8a914f1fa5e4a90fd5a23f170ed","coverage":[{"denominator":89,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":89,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:28:14.907661Z","state":"measured"},{"denominator":104,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":104,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":15,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T16:51:24.717294Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T17:27:15.782237Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-08-06T16:51:24.717294Z","title":"Perceive anything: Recognize, explain, caption, and segment anything in images and videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.12441","last_updated":"2025-08-02T17:35:59Z","snapshot_observed_at":"2026-08-08T01:34:56.747369Z","submitted_at":"2025-07-16T17:28:19Z","title":"Describe Anything Model for Visual Question Answering on Text-rich Images","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T16:51:24.717294Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2507.12441"},"observation_digest":"sha256:aec02b027cc8c53a71ddbaf44a4c2ff069c2e8e49513a6ebdee66c1de3868991","observation_id":"1309521c-a60a-444f-ba0b-13861ed211f6","resolution":{"observed_at":"2026-08-06T16:51:24.717294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-08-05T20:29:11.675850Z","title":"Lin et al","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.10955","last_updated":"2025-08-14T07:25:45Z","snapshot_observed_at":"2026-08-09T12:54:37.629481Z","submitted_at":"2025-08-14T07:25:45Z","title":"Empowering Multimodal LLMs with External Tools: A Comprehensive Survey","version":1},"reference_index":293,"source":"arxiv_source","source_observed_at":"2026-08-05T20:29:11.675850Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2508.10955"},"observation_digest":"sha256:e04e12d6835e14043ef4ebe92409e6161dee94d7c8ef33b41f5066856513c556","observation_id":"54b3d678-10f0-4408-b6bb-06bdb6520ea5","resolution":{"observed_at":"2026-08-05T20:29:11.675850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-08-05T10:58:16.722899Z","title":"Perceive anything: Recognize, explain, caption, and segment anything in images and videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.722899Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:5721741a2cf34c00ac876035c56328737efb49b8a87e51b99c94c7686d30ca4e","observation_id":"2b0f3896-8ba2-43c2-ac5a-aa020adf157d","resolution":{"observed_at":"2026-08-05T10:58:16.722899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-08-03T18:15:05.829843Z","title":"Perceive anything: Recognize, explain, caption, and segment anything in images and videos.arXiv preprint arXiv:2506.05302, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.06276","last_updated":"2026-07-28T07:07:11Z","snapshot_observed_at":"2026-08-10T11:50:46.710342Z","submitted_at":"2025-12-06T03:59:21Z","title":"RefBench-PRO: Perceptual and Reasoning Oriented Benchmark for Referring Expression Comprehension","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T18:15:05.829843Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2512.06276"},"observation_digest":"sha256:dda0a15647e5f2cf4d3f9cc3b6db34017aa37e1f3ae3fcbca0ba49d8cf42f531","observation_id":"5bdff3cf-8f46-4932-b4a7-efc5fd55bbdc","resolution":{"observed_at":"2026-08-03T18:15:05.829843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2512.07348","last_updated":"2026-04-28T10:02:14Z","snapshot_observed_at":"2026-08-06T12:32:11.849043Z","submitted_at":"2025-12-08T09:40:11Z","title":"MICo-150K: A Comprehensive Dataset Advancing Multi-Image Composition","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-17T00:20:58.483350Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2512.07348"},"observation_digest":"sha256:8ebfd2bea4558afa59f65f8669a9cafaffd9ccf6e3bbbb892469f3e00c6a36fa","observation_id":"64af0091-4638-4aaa-8598-57b8cbcfa627","resolution":{"observed_at":"2026-05-17T00:21:23.379690Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2512.12675","last_updated":"2026-06-09T11:29:25Z","snapshot_observed_at":"2026-08-03T16:37:02.565200Z","submitted_at":"2025-12-14T12:58:19Z","title":"Scone: Bridging Composition and Distinction in Subject-Driven Image Generation via Unified Understanding-Generation Modeling","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T22:39:32.955779Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2512.12675"},"observation_digest":"sha256:b0bcc605c912bc27e622a4d106a6311620e583ed443a489de6a48715e5a1c099","observation_id":"f78f9f04-29ac-4021-89e9-4f639e9d59ea","resolution":{"observed_at":"2026-05-16T22:41:19.224758Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-08-03T16:37:05.085568Z","title":"Perceive anything: Recognize, explain, caption, and segment anything in images and videos.arXiv preprint arXiv:2506.05302, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.12675","last_updated":"2026-06-09T11:29:25Z","snapshot_observed_at":"2026-08-03T16:37:02.565200Z","submitted_at":"2025-12-14T12:58:19Z","title":"Scone: Bridging Composition and Distinction in Subject-Driven Image Generation via Unified Understanding-Generation Modeling","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-03T16:37:05.085568Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2512.12675"},"observation_digest":"sha256:8087d94139958ad0b1abc969ebe94818a525fe3a8776ef1f4a85fcd94eaf2cfd","observation_id":"59ec8268-c8fa-4276-9aee-c8c8bb744edd","resolution":{"observed_at":"2026-08-03T16:37:05.085568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2602.03151","last_updated":"2026-04-06T04:51:44Z","snapshot_observed_at":"2026-07-06T22:44:18.875334Z","submitted_at":"2026-02-03T06:06:35Z","title":"Enhancing Foundation VLM Robustness to Missing Modality: Scalable Diffusion for Bi-directional Feature Restoration","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-16T08:33:49.841678Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2602.03151"},"observation_digest":"sha256:e84532f4a1a47cd4c0a22bd23d6a8f7d83e97e6327495d650219febaf854cdaf","observation_id":"4d313a4e-0c65-4f54-b147-e665b3c3d17c","resolution":{"observed_at":"2026-05-16T08:37:37.229588Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2604.04707","last_updated":"2026-05-25T06:28:44Z","snapshot_observed_at":"2026-07-13T09:42:22.607962Z","submitted_at":"2026-04-06T14:19:48Z","title":"OpenWorldLib: A Unified Codebase and Definition of Advanced World Models","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-10T19:36:42.100191Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2604.04707"},"observation_digest":"sha256:d94942d9460fb4818b79c8a17b8b89dc6fd955a949c8f7f7bf9a9fc2631e22b1","observation_id":"728775e5-abdc-40d7-84b4-a2b3a515999e","resolution":{"observed_at":"2026-05-10T22:45:49.048227Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-13T09:42:23.808691Z","title":"Perceive anything: Recognize, explain, caption, and segment anything in images and videos, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.04707","last_updated":"2026-05-25T06:28:44Z","snapshot_observed_at":"2026-07-13T09:42:22.607962Z","submitted_at":"2026-04-06T14:19:48Z","title":"OpenWorldLib: A Unified Codebase and Definition of Advanced World Models","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-07-13T09:42:23.808691Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2604.04707"},"observation_digest":"sha256:62500f5ee5cf65aab897fb9fc81f486a689f6c9df1f6a874ab36e43b2f7286b5","observation_id":"87f48f3a-8a30-4c2e-baec-3acd0e31e9bc","resolution":{"observed_at":"2026-07-13T09:42:23.808691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2604.11789","last_updated":"2026-04-20T14:38:53Z","snapshot_observed_at":"2026-08-09T05:10:13.009841Z","submitted_at":"2026-04-13T17:55:02Z","title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","version":2},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-05-10T15:35:37.095627Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2604.11789"},"observation_digest":"sha256:3b7c74b0e25d7d0e7be6c1bfdd6810905d3e792cd681af22b98fc722e67a86de","observation_id":"c25c039a-fdd0-4fdc-a0c6-8609bd93d72d","resolution":{"observed_at":"2026-05-11T10:11:06.525843Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2605.16903","last_updated":"2026-05-16T09:28:46Z","snapshot_observed_at":"2026-07-06T23:27:57.805018Z","submitted_at":"2026-05-16T09:28:46Z","title":"WOW-Seg: A Word-free Open World Segmentation Model","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-19T21:23:14.311122Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2605.16903"},"observation_digest":"sha256:6fda39c9feecab270a51f93b85e6eaaf33f5ced0603da54b306ac455a8c9917c","observation_id":"fecadecc-bf3c-4c06-850e-e596473f3e66","resolution":{"observed_at":"2026-05-19T21:27:48.052668Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2605.18018","last_updated":"2026-05-18T08:09:37Z","snapshot_observed_at":"2026-07-06T23:28:55.079333Z","submitted_at":"2026-05-18T08:09:37Z","title":"See What I Mean: Aligning Vision and Language Representations for Video Fine-grained Object Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-20T12:10:54.874012Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2605.18018"},"observation_digest":"sha256:6ff25a1d60ca0e2652b82d578474a7ea9da6b71faf7d14fd918b296d907d21ce","observation_id":"d1ce6cc2-ff2d-43fa-bcdf-2036aa9755d6","resolution":{"observed_at":"2026-05-20T12:13:16.316928Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":114,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:11171556ade6ea7bf6a138332fe7d8a64d560d56bc11175427235136b6d09a6c","observation_id":"694d2d02-f883-4042-9944-73eace34e136","resolution":{"observed_at":"2026-07-02T17:27:15.783743Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-11T21:31:04.773453Z","title":"Perceive anything: Recognize, explain, cap- tion, and segment anything in images and videos,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04125","last_updated":"2026-07-05T05:42:56Z","snapshot_observed_at":"2026-08-09T01:44:53.560742Z","submitted_at":"2026-07-05T05:42:56Z","title":"FRFDet: Efficient UAV Small Object Detection with Symmetric Sampling and Scalable Fusion","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-07-11T21:31:04.773453Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2607.04125"},"observation_digest":"sha256:c7cf278275120a057c6e483b683cae3ee27e87fee7c9745eed5d5385237a84de","observation_id":"f067d0b1-ed72-4548-b86f-4a796ce217ed","resolution":{"observed_at":"2026-07-11T21:31:04.773453Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.05302/citation-record","integrity":"/paper/2506.05302/integrity","json":"/paper/2506.05302/citation-record.json","paper":"/paper/2506.05302"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:07.834487Z","title":"Mc-llava: Multi-concept personalized vision-language model.arXiv preprint arXiv:2411.11706, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:07.834487Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:fe597d4ec56651facd15a56b9b2506948c893347cb3c734fd81a3cfda668b69c","observation_id":"5e6c0141-9f53-4a7a-9fb3-8161ec7a3f0d","resolution":{"observed_at":"2026-08-07T10:28:07.834487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-07T10:28:07.986779Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:07.986779Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:ac0cce2d3e87b18a00d55ab3cf8882099dc097216ee4f113cd28213d73518041","observation_id":"9caa0b70-e2a3-4d5d-a811-fa791f80c4c2","resolution":{"observed_at":"2026-08-07T10:28:07.986779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.106718Z","title":"Meteor: An automatic metric for mt evaluation with improved correlation with human judgments","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.106718Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:99ddac2c025fd8a73f2f5a54dcda2bb0badd6a0a1555e0f3a396f19f39cb25f5","observation_id":"b44229d7-89fb-458d-934d-5bdf41176fcd","resolution":{"observed_at":"2026-08-07T10:28:08.106718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.211037Z","title":"Abductive commonsense reasoning, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.211037Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:2543c5b9538f6b8370231bb6f67e7286c1045a3bf285adc80975b0782193b5f3","observation_id":"78c355f2-42bb-4a4e-886c-ad97de098a34","resolution":{"observed_at":"2026-08-07T10:28:08.211037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.339094Z","title":"Graph cuts in vision and graphics: Theories and applications","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.339094Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:6233559cd1dd35e65da7884cdb99280538cfb9734cca3d2763ff44b9e981e982","observation_id":"53317ec5-f5f6-4ddc-b81d-c892bb967dcc","resolution":{"observed_at":"2026-08-07T10:28:08.339094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.465002Z","title":"Scene-text oriented referring expression comprehension.IEEE Transactions on Multimedia, 25:7208–7221, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.465002Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:e0a5b9bad594283cba4d7efb469f0fabaf4911334fcd56a2170934c7c70f4c01","observation_id":"b074da60-bfa4-4d8e-9662-7c15f588d927","resolution":{"observed_at":"2026-08-07T10:28:08.465002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.647907Z","title":"Activitynet: A large-scale video benchmark for human activity understanding","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.647907Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:7c9e6e2b7505b0961f3f900c14c209a8918a65e1cd54f8f34f60a273cdc0b0bf","observation_id":"b8625921-7c21-4aa7-9a75-856e626b190f","resolution":{"observed_at":"2026-08-07T10:28:08.647907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.783473Z","title":"Vip-llava: Making large multimodal models understand arbitrary visual prompts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.783473Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:34a8ed51e4aadf05eb4b051906447ee5025f1b82a65dd78bb2cfcbf3f17db9f9","observation_id":"6d1cb01a-b711-45b9-835b-124c53b2aa96","resolution":{"observed_at":"2026-08-07T10:28:08.783473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.952935Z","title":"Active contours without edges.IEEE Transactions on image processing, 10(2):266–277, 2001","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.952935Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:62ce299099105143964f45911383f54cf11f093f2ac2b9905e698ce70de80c09","observation_id":"9dac110d-8043-47ee-ae47-2bbdb3a948b2","resolution":{"observed_at":"2026-08-07T10:28:08.952935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:09.089611Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.089611Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:2b22d8dd4130a0f90dc53e4442244056b7e3e1c21ce14c593bd2bbf4aaf1d2fc","observation_id":"8a0a7662-5adb-4947-9e25-a5fee112ded4","resolution":{"observed_at":"2026-08-07T10:28:09.089611Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:09.157074Z","title":"Videollm-online: Online video large language model for streaming video","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.157074Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:d60ee7f98bbf09e6cc0f4f3f7a743560e00f22fa7234b34e8603a0b05c58fac0","observation_id":"b362ae58-7c9a-44a4-9249-8f628c09b56a","resolution":{"observed_at":"2026-08-07T10:28:09.157074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15195","last_updated":"2023-07-03T16:08:00Z","snapshot_observed_at":"2026-07-06T15:47:07.545213Z","submitted_at":"2023-06-27T04:31:52Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15195","snapshot_observed_at":"2026-08-07T10:28:09.209098Z","title":"Shikra: Unleashing multimodal llm’s referential dialogue magic.arXiv preprint arXiv:2306.15195, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.209098Z"},"links":{"cited_paper":"/paper/2306.15195","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:6c9904f4b2d790d7ef96956da5a796be600a43375a0d7ca60f43d96b80c200a0","observation_id":"3d57f293-dc7e-400f-8b59-ed0e79f1b019","resolution":{"observed_at":"2026-08-07T10:28:09.209098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06558","last_updated":"2023-05-11T04:33:08Z","snapshot_observed_at":"2026-08-03T19:49:16.800693Z","submitted_at":"2023-05-11T04:33:08Z","title":"Segment and Track Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06558","snapshot_observed_at":"2026-08-07T10:28:09.287312Z","title":"Segment and track anything.arXiv preprint arXiv:2305.06558, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.287312Z"},"links":{"cited_paper":"/paper/2305.06558","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:d525b9c0ebb8215446ed1b931451d4d8f98cb3cd15b03d1af9e61fa06d70a650","observation_id":"917015bf-08bc-43be-87e3-d8b9bc2d7aa5","resolution":{"observed_at":"2026-08-07T10:28:09.287312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:22.068971Z","title":"Total-text: A comprehensive dataset for scene text detection and recognition, 2017","venue":null,"work_id":"2e4b2f07-784c-4359-aeec-e3bb895cf44f","year":2017},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.366044Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:810119e28ae42e4f6d9cafb9be3ae6baafdf746553434b331ca5f356efdcde64","observation_id":"000a4c48-a6f7-46bc-8812-11e99f4309c4","resolution":{"observed_at":"2026-08-07T10:28:22.167513Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:21.866232Z","title":"V ocabulary-free image classification, 2024","venue":null,"work_id":"36727d25-b59d-477b-b169-4d20aa55642f","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.461655Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:66bc93587003163b03497ded867a2f75960970b7d755a32923c0cff076de33d7","observation_id":"9e559b13-e1bc-4755-b972-952d73d0b449","resolution":{"observed_at":"2026-08-07T10:28:21.982352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:21.299284Z","title":"Online action detection","venue":null,"work_id":"c3094ac9-ef89-4a97-a1f8-95a59bbee56e","year":2016},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.531677Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:4471fd388f1b245eed78613ce0ab9fa65f05ff463e7c861e45ee45fe51f98e75","observation_id":"6bccd78c-1e66-4279-9cda-1f1ba90c107c","resolution":{"observed_at":"2026-08-07T10:28:21.636264Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:21.069853Z","title":"Mevis: A large-scale benchmark for video segmentation with motion expressions, 2023","venue":null,"work_id":"0ecd032d-3ad7-4044-af6e-7754b58a032d","year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.622757Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:0ae39f33af876669f49bbd50f9961817a314aa76e0e64a536e8a8dc4d350cc1f","observation_id":"90bd38f4-2b46-4711-8c71-cd433798ce8c","resolution":{"observed_at":"2026-08-07T10:28:21.160819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.957306Z","title":"Actor and action video segmentation from a sentence","venue":null,"work_id":"495a8a83-683d-4dfb-ae5f-c51f637ccdb6","year":2018},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.710833Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:d9c25de21f60d2cfb8674b0062a96d528692040f3a9a961259fce6a79116421b","observation_id":"ec455c19-e304-4fd8-a4fd-c922bb6e03b0","resolution":{"observed_at":"2026-08-07T10:28:21.019625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.772810Z","title":"Icdar2017 robust reading challenge on coco-text","venue":null,"work_id":"2731c5b4-322f-463c-89b4-65c793652234","year":2017},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.778550Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:16a81ebd41c76f8e79e0318c2a218887dac877dda992179eedf19c8802d90b75","observation_id":"23fc4dd0-9b5d-4320-828c-a68112c74745","resolution":{"observed_at":"2026-08-07T10:28:20.857060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:09.831040Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.831040Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:3d5c2b6bd0bb0a11998aeadf2dc2c230114a2219772f51e60277388f52261fc1","observation_id":"93ee6ab1-f238-48e7-bb87-02d288870d6f","resolution":{"observed_at":"2026-08-07T10:28:09.831040Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.609377Z","title":"Regiongpt: Towards region understanding vision language model, 2024","venue":null,"work_id":"d8dceb18-3f0a-4506-b6a2-5b7d3201a8e1","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.953283Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:efb115925a007caf064d50a963c205e30313c8cc747d9c44310e23bfa97a4238","observation_id":"0fccdb72-31f9-4b8f-8922-eae20a984627","resolution":{"observed_at":"2026-08-07T10:28:20.698363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05643","last_updated":"2025-03-03T10:28:30Z","snapshot_observed_at":"2026-08-07T05:23:25.250251Z","submitted_at":"2024-10-08T02:46:30Z","title":"TRACE: Temporal Grounding Video LLM via Causal Event Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05643","snapshot_observed_at":"2026-08-07T10:28:10.068680Z","title":"Trace: Temporal grounding video llm via causal event modeling.arXiv preprint arXiv:2410.05643, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.068680Z"},"links":{"cited_paper":"/paper/2410.05643","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:618c385fcf10380d2e7f3e1df678a7dbbd52e40c03fe74acba943cf2001a29ec","observation_id":"8e817e83-8877-43fe-bf03-842cf459c6cb","resolution":{"observed_at":"2026-08-07T10:28:10.068680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:10.160396Z","title":"Lvis: A dataset for large vocabulary instance segmentation, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.160396Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:6101e8431fca5e1e6a17242f59d2e86ff0f1ed63170cdf830fb2ea042d44f937","observation_id":"b1a95b82-b009-4ffe-87f4-a71ce956a7b2","resolution":{"observed_at":"2026-08-07T10:28:10.160396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.456966Z","title":"Synthetic data for text localisation in natural images, 2016","venue":null,"work_id":"ec608a97-f88f-40b4-99f0-05762fdb35fb","year":2016},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.245238Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:cc099b0c4e795c090f21cc26d6d971c749a1d5a770a3dfcff871858f92940729","observation_id":"bf17f7ba-2721-403b-921c-fcb3e4933269","resolution":{"observed_at":"2026-08-07T10:28:20.526290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.269637Z","title":"Omni-rgpt: Unifying image and video region-level understanding via token marks, 2025","venue":null,"work_id":"a86bd877-3e47-45c6-8c99-009f463efcd8","year":2025},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.341914Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:b0ec6b1eaf2ff550d2bf72eb8c2a1dcbb2fe7be064e72a17a7f5c63025044de8","observation_id":"9f6b5a53-9742-454a-9744-3d2693faf534","resolution":{"observed_at":"2026-08-07T10:28:20.368507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.079593Z","title":"Segment and caption anything","venue":null,"work_id":"7fdc2332-e986-4f08-8e9e-43e5d52d1754","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.442313Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:3fcdbb10d47787a7a5786f1b2bd995112a5293ac1101ba9162d09d0aa2c3e100","observation_id":"329c1ee0-6d7c-4793-b219-11a137c499d1","resolution":{"observed_at":"2026-08-07T10:28:20.174249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T10:28:10.544501Z","title":"Gpt-4o system card.arXiv preprint arXiv:2410.21276, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.544501Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:810db54c3a6fef0fb5526ab3bef1a53929e54321d7e44f594d64c2f12d5a2646","observation_id":"8f5ee2b8-10b5-4d15-9577-c1b75af42a49","resolution":{"observed_at":"2026-08-07T10:28:10.544501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:10.660851Z","title":"Visual prompt tuning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.660851Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:505dd799dd0954ceb9cc1d66099abf3382ddbf6b8fe4ac4e5ad270d7b0140789","observation_id":"da1fde41-df61-47a4-b776-fe83df5a39a0","resolution":{"observed_at":"2026-08-07T10:28:10.660851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.18363","last_updated":"2025-03-11T14:19:42Z","snapshot_observed_at":"2026-08-09T11:37:16.443400Z","submitted_at":"2024-11-27T14:11:10Z","title":"ChatRex: Taming Multimodal LLM for Joint Perception and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.18363","snapshot_observed_at":"2026-08-07T10:28:10.735473Z","title":"Chatrex: Taming multimodal llm for joint perception and understanding.arXiv preprint arXiv:2411.18363, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.735473Z"},"links":{"cited_paper":"/paper/2411.18363","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:4976a526aaf40ef5335dad5f5cde5de42ae4990bbd584678dec2771b268ed880","observation_id":"7e4fafc4-c40b-4f2c-abba-fae1b630d798","resolution":{"observed_at":"2026-08-07T10:28:10.735473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.915868Z","title":"Icdar 2015 competition on robust reading","venue":null,"work_id":"c90a6807-829d-4411-842e-bbb58b1bdcbf","year":2015},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.822007Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:f503be9bcb2d5761c7c3772150426b981ee942a822dab431b60993cc1bb36875","observation_id":"0320ae6a-736d-466a-9137-320cf8b0beb0","resolution":{"observed_at":"2026-08-07T10:28:19.977543Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.764195Z","title":"Icdar 2013 robust reading competition","venue":null,"work_id":"2dcae961-b5e0-4a8b-8221-d9e14d44f3bf","year":2013},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.902744Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:6c16a6f8c2fab9e748ce67d9fedcc153f99c8aecd6da781e2ce3da30e48e4547","observation_id":"79fca956-3d36-4d13-8c15-fc1fce4ea5a4","resolution":{"observed_at":"2026-08-07T10:28:19.819618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:10.990032Z","title":"Referitgame: Referring to objects in photographs of natural scenes","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.990032Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:9e96f2c58c9b47bff170c26a5ac18f460660685e41c51982b494ace0f370e708","observation_id":"75fff697-834b-493d-9c0e-ebd32d1da19c","resolution":{"observed_at":"2026-08-07T10:28:10.990032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:11.093403Z","title":"Segment anything in high quality.Advances in Neural Information Processing Systems, 36:29914–29934, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.093403Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:5f642b75133d138ca8bde871a896407f842dc50a11b4536077cbe82348b1e0ce","observation_id":"7dd22ac1-8d1d-418a-872d-1c1d175764ca","resolution":{"observed_at":"2026-08-07T10:28:11.093403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.02643","last_updated":"2023-04-05T17:59:46Z","snapshot_observed_at":"2026-08-08T05:14:59.435033Z","submitted_at":"2023-04-05T17:59:46Z","title":"Segment Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.02643","snapshot_observed_at":"2026-08-07T10:28:11.178747Z","title":"Segment anything.arXiv preprint arXiv:2304.02643, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.178747Z"},"links":{"cited_paper":"/paper/2304.02643","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:68f239e03d4abd9b50f913a4f025a9fd5c07cd698e3aba59187c3832f6e7ae15","observation_id":"9594d68a-989f-41aa-835f-5303e38fd1d0","resolution":{"observed_at":"2026-08-07T10:28:11.178747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.655205Z","title":"Openimages: A public dataset for large-scale multi-label and multi-class image classification.Dataset available from https://github","venue":null,"work_id":"cef24270-7ec2-44ae-ae98-f91c9c02d4fe","year":2017},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.254087Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:7c48d26a13a2a982c97a559b55972d4c07e7a45eebbd20d4202942f0fe3df2e5","observation_id":"02a13c13-2da8-45f8-bc92-16ded190fae2","resolution":{"observed_at":"2026-08-07T10:28:19.690256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.491412Z","title":"Shamma, Michael S","venue":null,"work_id":"1b7d255e-6025-4d3a-90d2-1a93a1ade2fe","year":2016},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.338922Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:41d83ec1adf9117c55b2fb44f90dff9716956b9a5910a401f655c94020f223b9","observation_id":"dbc8b219-ba86-4eeb-a622-83fad787060b","resolution":{"observed_at":"2026-08-07T10:28:19.558320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.325530Z","title":"Beyond mot: Semantic multi-object tracking","venue":null,"work_id":"78027468-0052-4c2d-a810-1424276215c0","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.411250Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:aabf9f45ca3cd37dbbf1b623fc10cddccaa139db28e03f91f36639f970eeb0cb","observation_id":"2343811e-b4e8-4161-99c3-e1eed22d929b","resolution":{"observed_at":"2026-08-07T10:28:19.413523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16072","last_updated":"2025-04-22T17:51:41Z","snapshot_observed_at":"2026-08-07T16:00:09.944765Z","submitted_at":"2025-04-22T17:51:41Z","title":"Describe Anything: Detailed Localized Image and Video Captioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16072","snapshot_observed_at":"2026-08-07T10:28:11.485753Z","title":"Describe anything: Detailed localized image and video captioning.arXiv preprint arXiv:2504.16072, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.485753Z"},"links":{"cited_paper":"/paper/2504.16072","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:3a6dcd55da9983050665bb4e9c16ff23b9d114ae81722e6602699b074006727f","observation_id":"3de15e83-db5f-4f6d-8e47-8de78f50c331","resolution":{"observed_at":"2026-08-07T10:28:11.485753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:11.573027Z","title":"Rouge: A package for automatic evaluation of summaries","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.573027Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:9b146c304f597a5f761e99e7296bcb1464e95a6ac99ceffedd1066c2f5da963e","observation_id":"a5acc589-816e-4212-9a41-05c03a5b85cc","resolution":{"observed_at":"2026-08-07T10:28:11.573027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:11.641278Z","title":"Lawrence Zitnick, and Piotr Dollár","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.641278Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:d44e8b9af7d1df52038814056466fae186db2da500c1b602617ff6f507456fec","observation_id":"77c18e82-2491-4fb3-a229-e0f3415129a5","resolution":{"observed_at":"2026-08-07T10:28:11.641278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-09T18:37:19.036481Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-08-07T10:28:11.715662Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.715662Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:dc7c79731f7c058871d0fdef0f0971024e186b5eef1614d63106c686528df8e8","observation_id":"734693f4-cdbe-494d-a851-2c3d8c130a24","resolution":{"observed_at":"2026-08-07T10:28:11.715662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-07T10:28:11.759586Z","title":"Deepseek-v3 technical report.arXiv preprint arXiv:2412.19437, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.759586Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:5053b5c510d7451ba4f9b559fdf409376bbd1a1506b1ee18ceba23e4c2723516","observation_id":"af9be36d-cf66-4a61-ae61-4bb17b033d4f","resolution":{"observed_at":"2026-08-07T10:28:11.759586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:11.842353Z","title":"Gres: Generalized referring expression segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.842353Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:a0a2a856e3d4178e8f42a5e3b0a2b8da0b0d37a5cbf14639d242d70fa02d257f","observation_id":"9b80ee9d-3b88-48a8-aa56-1bae4df04a53","resolution":{"observed_at":"2026-08-07T10:28:11.842353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-07T10:28:11.917393Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models.arXiv preprint arXiv:2402.05935, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.917393Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:e82de4a57af35614219b8001426cb3657c3b51e7559a8b29c178f073b320fab3","observation_id":"9626591a-cf6b-4f0e-85b6-962987fec048","resolution":{"observed_at":"2026-08-07T10:28:11.917393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:12.028467Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.028467Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:7066ecd7e7ec6c64018da943c937272b52fb4371bc88604034525ebea07daeb6","observation_id":"d60893a0-c1ee-47ba-8b40-24254752769b","resolution":{"observed_at":"2026-08-07T10:28:12.028467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14824","last_updated":"2023-07-13T05:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-26T16:32:47Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14824","snapshot_observed_at":"2026-08-07T10:28:12.090612Z","title":"Kosmos-2: Grounding multimodal large language models to the world.arXiv preprint arXiv:2306.14824, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.090612Z"},"links":{"cited_paper":"/paper/2306.14824","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:62261a731760de11ee2780457d702b61afcdddd28a46dcf75c2478170926ed83","observation_id":"2f5a9006-e068-4a88-9059-5b2f4069a2bb","resolution":{"observed_at":"2026-08-07T10:28:12.090612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.133837Z","title":"The 2017 davis challenge on video object segmentation, 2018","venue":null,"work_id":"d603845d-f0db-4bf6-a0da-45985af796fd","year":2017},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.161483Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:049687e72e5a344c4705ddc23c80f9f475667b851e8509ea5e5a7ff535ef9d82","observation_id":"40857708-b9b1-43ac-8f63-68cc2bbdbcff","resolution":{"observed_at":"2026-08-07T10:28:19.218820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.936646Z","title":"Artemis: Towards referential understanding in complex videos","venue":null,"work_id":"a400922d-ae93-469e-90df-99ef04627912","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.252891Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:d7133b771912b594f365cfed62ee894c9fa8b83cbe3d989e371af17aafba4d51","observation_id":"dca6c900-6108-4828-b8be-4fda60a8bde5","resolution":{"observed_at":"2026-08-07T10:28:19.045839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.765741Z","title":"Artemis: Towards referential understanding in complex videos, 2024","venue":null,"work_id":"b2a008eb-7923-4ad4-b294-ecc9bd06fea9","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.318885Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:17029f6ac37fc9d0cea4ccff163d7fba27491ce2fc09d2f07e78b531cfe9722a","observation_id":"0411d034-bbf6-4397-8a7b-2527829345fb","resolution":{"observed_at":"2026-08-07T10:28:18.841719Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.606462Z","title":"Paco: Parts and attributes of common objects, 2023","venue":null,"work_id":"5cd31bd2-06df-477a-a395-a5070ad0bba9","year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.398651Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:6708d85673884537c0ee0e8585adb4f9f11c0543ae4a7c13de68c4e1f0eb56db","observation_id":"615de475-4912-4b52-843a-076e6b94c72f","resolution":{"observed_at":"2026-08-07T10:28:18.669927Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:12.438893Z","title":"Anwer, Erix Xing, Ming-Hsuan Yang, and Fahad S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.438893Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:2bb7eddc95b0254445ccb317c8a8defe91886f764fb16c763d8c91dd015a5068","observation_id":"3b52654c-8e87-4765-9961-b934d46bc038","resolution":{"observed_at":"2026-08-07T10:28:12.438893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-07T10:28:12.549714Z","title":"Sam 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.549714Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:865053ee7fca060de25d9a2aff1a15569e80407589f94dca7d7b4dd49679fc3a","observation_id":"80503a7d-ce6a-4a73-abf5-43bb8717d469","resolution":{"observed_at":"2026-08-07T10:28:12.549714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:12.604590Z","title":"Sam 2: Segment anything in images and videos, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.604590Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:739d0d64b437090008c5ad24e1cd373b868c1f82203290aa2d2511aa97330617","observation_id":"77d3f82b-7679-4289-a451-10f39f150192","resolution":{"observed_at":"2026-08-07T10:28:12.604590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14159","last_updated":"2024-01-25T13:12:09Z","snapshot_observed_at":"2026-07-06T17:20:25.138890Z","submitted_at":"2024-01-25T13:12:09Z","title":"Grounded SAM: Assembling Open-World Models for Diverse Visual Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.14159","snapshot_observed_at":"2026-08-07T10:28:12.695638Z","title":"Grounded sam: Assembling open-world models for diverse visual tasks.arXiv preprint arXiv:2401.14159, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.695638Z"},"links":{"cited_paper":"/paper/2401.14159","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:c6a5998c352a7b276b5e2013fcdaab6be72c2e6b32fb5c712c97abd126aeb250","observation_id":"dc7a87ef-9de9-4a0f-a155-21f4c3590af6","resolution":{"observed_at":"2026-08-07T10:28:12.695638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:12.752865Z","title":"Objects365: A large-scale, high-quality dataset for object detection","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.752865Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:5622fe553fa0cd9eebf287f5a5ef9eec621e57c2a4f18f02373dcb7cca37e213","observation_id":"53fdc3f9-85b0-41cd-9010-687b9512139e","resolution":{"observed_at":"2026-08-07T10:28:12.752865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.433001Z","title":"Textocr: Towards large-scale end-to-end reasoning for arbitrary-shaped scene text, 2021","venue":null,"work_id":"1d599850-450a-4a0d-be75-bc955840ffcb","year":2021},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.830981Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:f744727a3d0b7868b4b132fa40de83b8057342676b05c8271146468aff98913d","observation_id":"bd84cbc5-8476-4f71-9d32-a5ca9a7245e6","resolution":{"observed_at":"2026-08-07T10:28:18.501128Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.286911Z","title":"Icdar 2019 competition on large-scale street view text with partial labeling – rrc-lsvt, 2019","venue":null,"work_id":"74117015-d431-445a-b4f5-a171cd857834","year":2019},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.899875Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:1fcf7e2b3c64750629408ead45bb11459308522282b23cea43186b9481266d05","observation_id":"003298e2-5e3e-4286-a1da-025dd32bb251","resolution":{"observed_at":"2026-08-07T10:28:18.357067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.116563Z","title":"Human-centric spatio-temporal video grounding with visual transformers, 2021","venue":null,"work_id":"ca513442-32e7-4c40-af91-f7bb06e4638f","year":2021},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.974590Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:bfaaffc989f63371ed9946768721ebe03b43f057ddb5cbff5dbc1f21c446c953","observation_id":"97ae4dcd-4b1e-4f1b-90fd-60373548ef85","resolution":{"observed_at":"2026-08-07T10:28:18.211229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.951510Z","title":"Human- centric spatio-temporal video grounding with visual transformers.IEEE Transactions on Circuits and Systems for Video Technology, 32(12):8238–8249, 2021","venue":null,"work_id":"d65ec784-2fdf-4b46-bf8b-301e2a2fdf08","year":2021},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.075724Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:a23daa2a957d39e757f47956f047c206b3699cd4cc94d54713f820ab82551c39","observation_id":"f81eebbf-4a2b-4617-958f-4e2a0ea98be1","resolution":{"observed_at":"2026-08-07T10:28:18.042690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:13.106006Z","title":"Cider: Consensus-based image description evaluation","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.106006Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:6bc44e6f96ccd4c0cd2180d79ecac812cbbd3b5e9a1524916ce38d2f24032012","observation_id":"a6c7bb8a-736d-4403-b5a3-847e4dbdc50a","resolution":{"observed_at":"2026-08-07T10:28:13.106006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.797415Z","title":"Coco-text: Dataset and benchmark for text detection and recognition in natural images, 2016","venue":null,"work_id":"2a8f37c0-03de-4cf4-a928-81cb1d7a7a15","year":2016},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.197387Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:a47ae8ee6f3bf2fdb1801677a7466907c3ee9dee0fa97bc80f7eeda216e447a8","observation_id":"6538d8ce-016e-4755-9409-05c67de2a623","resolution":{"observed_at":"2026-08-07T10:28:17.871637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.624767Z","title":"Elysium: Exploring object-level perception in videos via mllm, 2024","venue":null,"work_id":"4b7741c9-b5a3-40b8-97d0-01e2b1fa78bc","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.272446Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:a9f007dad52359bef4e794bef0d94d7fd99b07819762a1f43be0b4c9a4a45e91","observation_id":"2be0f5f2-0406-4073-b936-c8820e72116c","resolution":{"observed_at":"2026-08-07T10:28:17.712210Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.462594Z","title":"Elysium: Exploring object-level perception in videos via mllm","venue":null,"work_id":"1bc77669-60ab-4e66-b07b-c6764de5ecf0","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.376288Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:4a33d940b827a614a6a18d28dad7050beda5cab35e6398c897c4d2fd83d51650","observation_id":"86242f7d-353d-4fce-9401-a217cb7e31d1","resolution":{"observed_at":"2026-08-07T10:28:17.544629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.316954Z","title":"Towards open-vocabulary video instance segmentation, 2023","venue":null,"work_id":"dae86712-c075-4f92-b220-cb7eb5c853a8","year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.408456Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:00c7fb9e8f648275f024da0521634ddd2ddeb312f0bda5bdafe714177d62ac0b","observation_id":"d27a146f-4d23-4b7a-8f8a-50063baddd19","resolution":{"observed_at":"2026-08-07T10:28:17.380342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:13.430957Z","title":"Git: A generative image-to-text transformer for vision and language, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.430957Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:548ab57aa227fe74a405f3ba2dc590a7a8f09ae6fc783700b34e912442db2605","observation_id":"24ea961e-be69-4b03-8336-d84a91f841d6","resolution":{"observed_at":"2026-08-07T10:28:13.430957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.170260Z","title":"V3det: Vast vocabulary visual detection dataset, 2023","venue":null,"work_id":"09b59949-2c9b-4b86-b272-062425adf1e3","year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.449740Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:2432787a319d525b12d74afeb13f26d941298e1efe698250260ef2b549b3e788","observation_id":"b4f51c45-574e-48be-87e2-b80093e76b7c","resolution":{"observed_at":"2026-08-07T10:28:17.243065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.02677","last_updated":"2023-07-06T13:47:21Z","snapshot_observed_at":"2026-08-09T15:56:42.519342Z","submitted_at":"2023-05-04T09:48:22Z","title":"Caption Anything: Interactive Image Description with Diverse Multimodal Controls","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.02677","snapshot_observed_at":"2026-08-07T10:28:13.474745Z","title":"Caption anything: Interactive image description with diverse multimodal controls.arXiv preprint arXiv:2305.02677, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.474745Z"},"links":{"cited_paper":"/paper/2305.02677","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:b2acf6c2dbb8a42aaf7ef9c5bc74213d7b671d6b91bac4d589e1bd5d909f165c","observation_id":"64243a39-bee3-4aca-b170-7f18f7ea9eea","resolution":{"observed_at":"2026-08-07T10:28:13.474745Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01907","last_updated":"2023-08-03T17:59:47Z","snapshot_observed_at":"2026-07-06T16:02:15.266899Z","submitted_at":"2023-08-03T17:59:47Z","title":"The All-Seeing Project: Towards Panoptic Visual Recognition and Understanding of the Open World","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.01907","snapshot_observed_at":"2026-08-07T10:28:13.496607Z","title":"The all-seeing project: Towards panoptic visual recognition and understanding of the open world.arXiv preprint arXiv:2308.01907, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.496607Z"},"links":{"cited_paper":"/paper/2308.01907","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:8401e7549d392df55be0848d8c876a27ba8947eeb2d4cf2ac3a3248f36335619","observation_id":"7b4824b5-1247-46f0-9acb-6b6f537991e6","resolution":{"observed_at":"2026-08-07T10:28:13.496607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.973287Z","title":"Grit: A generative region-to-text transformer for object understanding","venue":null,"work_id":"d2c5e61b-a93f-465a-a2c1-88f264021de9","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.564740Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:428ae2621a1b9817a66c1544621e90ed50191a6c4e1bca1d0262ad26f8024489","observation_id":"ce98c32a-3bee-4a17-b336-b03bff860f17","resolution":{"observed_at":"2026-08-07T10:28:17.065248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:13.586867Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model for hundreds of vision-language tasks.Advances in Neural Information Processing Systems, 37:69925–69975, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.586867Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:73e03ecaec620991113cfcb9f593a844ea328139cf7d70e17a5938d287b12087","observation_id":"feb50aa7-36ce-4526-9e51-e569d7c557cb","resolution":{"observed_at":"2026-08-07T10:28:13.586867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.789844Z","title":"Youtube-vos: A large-scale video object segmentation benchmark, 2018","venue":null,"work_id":"edfab5fb-0b7b-429c-bace-0f8638b8b022","year":2018},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.605063Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:2b58bb05e44cf0208faa5a46ac167e9f04f48e5663aa09df8f3ae6071343a929","observation_id":"4e4340d2-d0db-4de9-ba17-cf974d152d9f","resolution":{"observed_at":"2026-08-07T10:28:16.863847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T10:28:13.643014Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.643014Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:0124596bf84029cd739dad45e0614a7005cc3f2bbe0e39740ac6663954e4b952","observation_id":"6306acd4-37f1-40c5-8523-43cfcc258b31","resolution":{"observed_at":"2026-08-07T10:28:13.643014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.509176Z","title":"Vid2seq: Large-scale pretraining of a visual language model for dense video captioning, 2023","venue":null,"work_id":"f4162725-7b4c-4fdf-a5bf-622e5a972704","year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.676752Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:682f49f91671664abb0aef1f69ab1188b529b1d092bd9025f8044c34cb243250","observation_id":"bba35fb9-1d5f-43a0-9e47-1b2befb8714e","resolution":{"observed_at":"2026-08-07T10:28:16.586598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.352292Z","title":"Samurai: Adapting segment anything model for zero-shot visual tracking with motion-aware memory, 2024","venue":null,"work_id":"6b3b498b-a18c-44a0-a1b0-bbd98a4627fa","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.714771Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:d2d15fcfe35e8f43c7822a008d82cfb99ca0dadf34e59740986e9d7a090daf22","observation_id":"038aa238-4d5e-452e-897b-8f9fd415f96a","resolution":{"observed_at":"2026-08-07T10:28:16.410525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:13.738804Z","title":"Set-of-mark prompting unleashes extraordinary visual grounding in gpt-4v, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.738804Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:4886ed3501ca4113372fe6da8d26cad94fee6793090e4448afb1bc53c6c23d1f","observation_id":"6f24cf1c-e96e-498b-94ec-f38cf6b5d0a6","resolution":{"observed_at":"2026-08-07T10:28:13.738804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.203152Z","title":"Detecting texts of arbitrary orientations in natural images","venue":null,"work_id":"ac7158e4-9d3c-43f1-aad3-c04b04fc377a","year":2012},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.788564Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:dd99b1fe9470d085a7cbf031a26a3a3da16a1eff95c82efb5228197ac270b321","observation_id":"7b581093-f66b-4a18-8762-233b527d2abf","resolution":{"observed_at":"2026-08-07T10:28:16.269910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07704","last_updated":"2023-10-11T17:55:15Z","snapshot_observed_at":"2026-07-06T16:31:25.350087Z","submitted_at":"2023-10-11T17:55:15Z","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07704","snapshot_observed_at":"2026-08-07T10:28:13.842739Z","title":"Ferret: Refer and ground anything anywhere at any granularity.arXiv preprint arXiv:2310.07704, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.842739Z"},"links":{"cited_paper":"/paper/2310.07704","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:b5d92992f21b6d1d4c922172c70d0ef7f075304b260f93c25fdab57ab9505564","observation_id":"15915057-4c0d-42e2-9ead-e8bb545702d5","resolution":{"observed_at":"2026-08-07T10:28:13.842739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:13.904324Z","title":"Merlin: Empowering multimodal llms with foresight minds","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.904324Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:deb0f3f503fe66bbde8fd890240fe6e5e664c746cdba863b2493dc366778a1d8","observation_id":"54f1de86-fd8f-4e15-ac4e-e612520875a4","resolution":{"observed_at":"2026-08-07T10:28:13.904324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.007221Z","title":"Sa2va: Marrying sam2 with llava for dense grounded understanding of images and videos, 2025","venue":null,"work_id":"0ad46f26-a4fc-487d-8483-5195049792cb","year":2025},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.964588Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:dd7e4d474de425fe302192232a3c1417400dd33988628f456d3e160ea9eb8182","observation_id":"04667787-33fe-4d7a-b681-83be6aea276a","resolution":{"observed_at":"2026-08-07T10:28:16.098427Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:14.049510Z","title":"Osprey: Pixel understanding with visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.049510Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:57cbc5d961ea89c9f1b1f43852b908ad8048781c74346a971c62f0a412122f10","observation_id":"3532d1b8-250b-4955-bf83-c62da59b9e9e","resolution":{"observed_at":"2026-08-07T10:28:14.049510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00599","last_updated":"2025-03-25T08:10:15Z","snapshot_observed_at":"2026-08-08T18:38:08.024723Z","submitted_at":"2024-12-31T18:56:46Z","title":"VideoRefer Suite: Advancing Spatial-Temporal Object Understanding with Video LLM","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00599","snapshot_observed_at":"2026-08-07T10:28:14.117492Z","title":"Videorefer suite: Advancing spatial-temporal object understanding with video llm.arXiv preprint arXiv:2501.00599, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.117492Z"},"links":{"cited_paper":"/paper/2501.00599","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:e142fcadd184b5263878263903b3d944d247c88270492a2b13b09b4aed97ed04","observation_id":"60a624db-a6fa-43c6-aedc-ac269ea708c2","resolution":{"observed_at":"2026-08-07T10:28:14.117492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14289","last_updated":"2023-07-01T07:26:22Z","snapshot_observed_at":"2026-08-08T16:17:33.420400Z","submitted_at":"2023-06-25T16:37:25Z","title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14289","snapshot_observed_at":"2026-08-07T10:28:14.194914Z","title":"Faster segment anything: Towards lightweight sam for mobile applications.arXiv preprint arXiv:2306.14289, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.194914Z"},"links":{"cited_paper":"/paper/2306.14289","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:9097ffc1f79866de9f34777be5cfa317211ed70b9f80f0b7757921bd94f227a8","observation_id":"86547f57-af2f-49a4-aa87-4cd8e57a6fa5","resolution":{"observed_at":"2026-08-07T10:28:14.194914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07973","last_updated":"2024-04-11T17:56:05Z","snapshot_observed_at":"2026-08-07T09:16:48.366534Z","submitted_at":"2024-04-11T17:56:05Z","title":"Ferret-v2: An Improved Baseline for Referring and Grounding with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07973","snapshot_observed_at":"2026-08-07T10:28:14.279027Z","title":"Ferret-v2: An improved baseline for referring and grounding with large language models.arXiv preprint arXiv:2404.07973, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.279027Z"},"links":{"cited_paper":"/paper/2404.07973","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:36a712c5fb4412a18a60d38b65361412f2cbec43d6d2e03fd41dde668021f3c0","observation_id":"ae642888-c3de-4cbe-aa24-3161505b0301","resolution":{"observed_at":"2026-08-07T10:28:14.279027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:15.817591Z","title":"Gpt4roi: Instruction tuning large language model on region-of-interest, 2025","venue":null,"work_id":"d50fe2a2-c302-4c6a-8846-53f71aa8c351","year":2025},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.380692Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:75d2d2374ab7e016b3043cda93ac76c20495ba5fc51fab8908c3510d8f48235d","observation_id":"e377b8a5-dc62-43a2-b95e-5499da6488b8","resolution":{"observed_at":"2026-08-07T10:28:15.913568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:15.675110Z","title":"Where does it exist: Spatio-temporal video grounding for multi-form sentences","venue":null,"work_id":"f0c31e77-381f-4875-8b03-6580131db20d","year":2020},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.500909Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:01432d6f6324d8274831ab0b2396787c31e5b58ee6e9828efa0115ecabacf9d3","observation_id":"614da7e3-d07b-487c-875c-d939bf00e4fa","resolution":{"observed_at":"2026-08-07T10:28:15.744187Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09474","last_updated":"2023-07-18T17:56:06Z","snapshot_observed_at":"2026-08-09T11:47:54.530329Z","submitted_at":"2023-07-18T17:56:06Z","title":"ChatSpot: Bootstrapping Multimodal LLMs via Precise Referring Instruction Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09474","snapshot_observed_at":"2026-08-07T10:28:14.565949Z","title":"Chatspot: Bootstrapping multimodal llms via precise referring instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.565949Z"},"links":{"cited_paper":"/paper/2307.09474","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:68a01b98983473946324083f7d747e2330598765490b23e691f625f5f168dc58","observation_id":"ccdc906f-94a4-4d76-bf34-91859924d5ba","resolution":{"observed_at":"2026-08-07T10:28:14.565949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.12156","last_updated":"2023-06-21T10:08:29Z","snapshot_observed_at":"2026-07-06T15:45:01.961896Z","submitted_at":"2023-06-21T10:08:29Z","title":"Fast Segment Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.12156","snapshot_observed_at":"2026-08-07T10:28:14.704495Z","title":"Fast segment anything.arXiv preprint arXiv:2306.12156, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.704495Z"},"links":{"cited_paper":"/paper/2306.12156","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:8a9440d87cd339a7c0fcc686db0696ba84152dd089dd6ab68a71c21bcb193fa0","observation_id":"037adc13-83fb-4e76-a27e-08af8718a73b","resolution":{"observed_at":"2026-08-07T10:28:14.704495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:15.491527Z","title":"Controlcap: Controllable region-level captioning","venue":null,"work_id":"6bc19424-1ffe-4ac9-806f-accdf1cc1fb9","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.776221Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:0c2ce84702ddfdd00cf80e836c29a0773ebf888a4b47299d1930ab071f6a94b8","observation_id":"1b5a5567-5bb8-4335-a8ee-4977daff2dd2","resolution":{"observed_at":"2026-08-07T10:28:15.590718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:15.323383Z","title":"Streaming dense video captioning","venue":null,"work_id":"5bab4f21-a0ce-4e4b-8ca1-e4ee75dcc2ad","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.907661Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:21fba3ac4992c6560d553229a5f828290af58705893fb684bb06e4ae8adec07d","observation_id":"103c9006-1579-4b31-b1f1-df970f5937e1","resolution":{"observed_at":"2026-08-07T10:28:15.393823Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T01:44:49.494240Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos"},"reference_resolution":{"displayed":89,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":51,"verified_exact":0,"verified_fuzzy":38},"total_outbound_references":89},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 89 of 89 outbound references and 15 inbound Pith citation observations for arXiv:2506.05302."}