{"as_of":"2026-08-10T05:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fcb51861c370481d0492460940a6ec621abdcbfa86fd19e795f81a2cd48ab8ae","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":21,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":21,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":21,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T20:36:27.109914Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T15:58:38.244809Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2401.14159","last_updated":"2024-01-25T13:12:09Z","snapshot_observed_at":"2026-07-06T17:20:25.138890Z","submitted_at":"2024-01-25T13:12:09Z","title":"Grounded SAM: Assembling Open-World Models for Diverse Visual Tasks","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-11T06:20:15.656356Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2401.14159"},"observation_digest":"sha256:984f5e1b04290c566eabdbf07abee68ec61450d7edeb02f3a6c88c5bc1293477","observation_id":"f77e605a-4257-48f1-bf6e-74856ab78195","resolution":{"observed_at":"2026-05-11T06:20:16.231345Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2402.00253","last_updated":"2024-05-06T01:10:01Z","snapshot_observed_at":"2026-07-06T17:23:26.913214Z","submitted_at":"2024-02-01T00:33:21Z","title":"A Survey on Hallucination in Large Vision-Language Models","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-13T22:10:10.186950Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2402.00253"},"observation_digest":"sha256:4ba5102a6b5a268c68c546e917af4a5da1e82a5f321fde23baa493a2e6d57968","observation_id":"2bb2d1cd-a897-4e51-8ef6-a3cfab578efc","resolution":{"observed_at":"2026-05-13T22:10:10.262114Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-09T20:36:27.109914Z","title":"Recognize anything: A strong image tagging model,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.19331","last_updated":"2025-01-31T17:31:19Z","snapshot_observed_at":"2026-08-10T02:27:20.743382Z","submitted_at":"2025-01-31T17:31:19Z","title":"Consistent Video Colorization via Palette Guidance","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-09T20:36:27.109914Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2501.19331"},"observation_digest":"sha256:f6efab50a387426708ed5c1c89331335c9967c3839a3547a91fa4a8af93ba9e5","observation_id":"4351ae15-7939-46f5-93d9-5f9415e714d2","resolution":{"observed_at":"2026-08-09T20:36:27.109914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-09T11:52:15.003082Z","title":"Recognize anything: A strong image tagging model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.02548","last_updated":"2025-04-14T18:27:02Z","snapshot_observed_at":"2026-08-10T03:02:09.680267Z","submitted_at":"2025-02-04T18:18:50Z","title":"Mosaic3D: Foundation Dataset and Model for Open-Vocabulary 3D Segmentation","version":2},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-08-09T11:52:15.003082Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2502.02548"},"observation_digest":"sha256:72b0869a7302f0fdee2b76e7decb56a73a8ad49e4d6a3f2d7be27eda8ae97116","observation_id":"545cf9cc-3c2d-4110-9fd7-fa9c111dc442","resolution":{"observed_at":"2026-08-09T11:52:15.003082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2503.07598","last_updated":"2025-03-11T06:44:25Z","snapshot_observed_at":"2026-07-06T20:50:05.070886Z","submitted_at":"2025-03-10T17:57:04Z","title":"VACE: All-in-One Video Creation and Editing","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-16T00:53:53.855965Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2503.07598"},"observation_digest":"sha256:143838c1fe80a9c37e66b191040561b1ebae7817d4abcfd905a3e75904867a4c","observation_id":"30e699ae-cadf-4b19-8586-7092962bdd79","resolution":{"observed_at":"2026-05-16T00:53:54.113310Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2503.07878","last_updated":"2026-04-29T20:30:35Z","snapshot_observed_at":"2026-08-03T14:12:26.604832Z","submitted_at":"2025-03-10T21:50:58Z","title":"A Woman with a Knife or A Knife with a Woman? Measuring Directional Bias Amplification in Image Captions","version":5},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-22T23:54:15.135595Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2503.07878"},"observation_digest":"sha256:d5a28fc3193751c2c8c23cbe7e2bbd7a02b0140a0b4f4ccbc15ad7700b47b1de","observation_id":"db08e31d-29fd-4975-a579-a9bb6949759d","resolution":{"observed_at":"2026-05-22T23:55:15.097067Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2504.17761","last_updated":"2025-07-31T05:05:45Z","snapshot_observed_at":"2026-07-06T21:14:24.007428Z","submitted_at":"2025-04-24T17:25:12Z","title":"Step1X-Edit: A Practical Framework for General Image Editing","version":5},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-11T14:36:41.467429Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2504.17761"},"observation_digest":"sha256:306ceffbca5e9048b8e804c012e55ee19a9e002c2b789d00c76f404fdf8071be","observation_id":"36865902-00ca-4044-89f6-bdc67c795e50","resolution":{"observed_at":"2026-05-11T14:36:41.924000Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-07T14:14:35.758398Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19569","last_updated":"2025-05-26T06:33:48Z","snapshot_observed_at":"2026-08-10T03:02:25.122420Z","submitted_at":"2025-05-26T06:33:48Z","title":"What You Perceive Is What You Conceive: A Cognition-Inspired Framework for Open Vocabulary Image Segmentation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T14:14:35.758398Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2505.19569"},"observation_digest":"sha256:6e4dbaee665ddc5d731faa84d4014827cf54bd3b5c3f4cb474980fb6cab7047c","observation_id":"35cdf453-fea3-4095-8941-1fcc73ad8d74","resolution":{"observed_at":"2026-08-07T14:14:35.758398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-07T05:03:16.194374Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08968","last_updated":"2025-06-10T16:41:33Z","snapshot_observed_at":"2026-08-07T13:57:35.713590Z","submitted_at":"2025-06-10T16:41:33Z","title":"ADAM: Autonomous Discovery and Annotation Model using LLMs for Context-Aware Annotations","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T05:03:16.194374Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2506.08968"},"observation_digest":"sha256:d30ea8109bcc752e78ec8162f5cf6c2d97066f37b5183648851c2a9ea6794d8f","observation_id":"baae690b-501f-4003-bd4b-b06b3901c93b","resolution":{"observed_at":"2026-08-07T05:03:16.194374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-06T16:58:56.201780Z","title":"Recognize Anything: A Strong Image Tagging Model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12123","last_updated":"2025-07-16T10:47:12Z","snapshot_observed_at":"2026-08-10T03:01:21.197103Z","submitted_at":"2025-07-16T10:47:12Z","title":"Open-Vocabulary Indoor Object Grounding with 3D Hierarchical Scene Graph","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T16:58:56.201780Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2507.12123"},"observation_digest":"sha256:9e145cc343cfe4c67b4f186dbf262082ee846f626632c83c819b14c5ddc20008","observation_id":"5f0f4907-fe50-4cc2-89aa-f235015a8176","resolution":{"observed_at":"2026-08-06T16:58:56.201780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-05T12:27:05.091735Z","title":"Recognize anything: A strong image tagging model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.01656","last_updated":"2025-09-01T17:57:49Z","snapshot_observed_at":"2026-08-09T16:17:17.533156Z","submitted_at":"2025-09-01T17:57:49Z","title":"Reinforced Visual Perception with Tools","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-05T12:27:05.091735Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2509.01656"},"observation_digest":"sha256:1cc40e64213c7d2d78c62be5392eb5b4ad15c4b4ee74b097f2140f7ef8798c25","observation_id":"fff8ffb9-0e83-4198-b399-5c03635fc775","resolution":{"observed_at":"2026-08-05T12:27:05.091735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2604.18562","last_updated":"2026-04-22T02:31:50Z","snapshot_observed_at":"2026-08-04T10:02:00.731719Z","submitted_at":"2026-04-20T17:49:22Z","title":"AnchorSeg: Language Grounded Query Banks for Reasoning Segmentation","version":3},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-10T05:10:44.608959Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2604.18562"},"observation_digest":"sha256:75f3a694046bcc33282d70ee4da9a5651cfec517af5ce77cd18320bf638551e1","observation_id":"be448d09-7cc2-4571-bb94-51e5a993bebc","resolution":{"observed_at":"2026-05-10T09:43:49.396778Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2604.19192","last_updated":"2026-04-23T06:35:35Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:59:36Z","title":"Empowering NPC Dialogue with Environmental Context Using LLMs and Panoramic Images","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-10T01:45:42.756787Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2604.19192"},"observation_digest":"sha256:19ed7dcbac02acbf8de2efce9cd1a8a65c8771143eb6bd6aaa0c25ab681f316b","observation_id":"a95c5bef-7906-4e3e-9d71-a8682a003bf3","resolution":{"observed_at":"2026-05-11T13:26:03.724298Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2604.21915","last_updated":"2026-04-23T17:57:28Z","snapshot_observed_at":"2026-07-06T23:08:23.939574Z","submitted_at":"2026-04-23T17:57:28Z","title":"Vista4D: Video Reshooting with 4D Point Clouds","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-09T22:07:10.070757Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2604.21915"},"observation_digest":"sha256:1bc06bd621f12edb91bb629271f04babab220160fccc97786ac152a33fc0a835","observation_id":"1bca43ee-80b4-4498-837c-8f35d9944663","resolution":{"observed_at":"2026-05-11T14:16:26.263787Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2605.07141","last_updated":"2026-05-08T02:20:40Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:20:40Z","title":"Qwen3-VL-Seg: Unlocking Open-World Referring Segmentation with Vision-Language Grounding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-11T02:35:57.843351Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2605.07141"},"observation_digest":"sha256:5a2cf81f3566eed941b122ef258b4bc8ad198255378daa0900f18c73eef71bf0","observation_id":"28b87019-9fae-4f06-9162-8828411f7cb8","resolution":{"observed_at":"2026-05-11T03:10:53.994295Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2605.21244","last_updated":"2026-05-20T14:33:13Z","snapshot_observed_at":"2026-08-07T04:02:33.199414Z","submitted_at":"2026-05-20T14:33:13Z","title":"SR-Ground: Image Quality Grounding for Super-Resolved Content","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-21T05:34:17.056685Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2605.21244"},"observation_digest":"sha256:f8baa290e8fab1a901c50e42946023d82ae422526a5054b5c5b691e721867b04","observation_id":"928a528b-2563-47d9-873f-f3b0b8833e91","resolution":{"observed_at":"2026-05-21T05:34:40.136791Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2606.01072","last_updated":"2026-06-05T06:32:49Z","snapshot_observed_at":"2026-08-02T21:41:54.386640Z","submitted_at":"2026-05-31T07:34:25Z","title":"Expanding Spatial and Temporal Context for Robotic Imitation Learning With Scene Graphs","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-06-28T17:24:02.843529Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2606.01072"},"observation_digest":"sha256:0c8eca971ac1f9f7de0e9e5df4db6965b449b44063d2522c6536129fb6be7a82","observation_id":"b0d85f2a-4280-4640-87f9-8a85c30d95fc","resolution":{"observed_at":"2026-07-01T21:16:13.673065Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2606.12849","last_updated":"2026-06-11T03:24:05Z","snapshot_observed_at":"2026-08-03T10:50:52.195357Z","submitted_at":"2026-06-11T03:24:05Z","title":"SemanticXR: Low Power and Real-time Queryable Semantic Mapping with an Object-Level Device-Cloud Architecture","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-27T06:13:49.396855Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2606.12849"},"observation_digest":"sha256:ae06d5db4e2b7b6c0affbc8dcd47d57e5f9ca63cc775917dd8eb257aadef0b77","observation_id":"37f6e3cc-df5a-4c54-afb8-df89e4b5768b","resolution":{"observed_at":"2026-07-03T15:58:38.248195Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2606.28592","last_updated":"2026-06-26T20:36:26Z","snapshot_observed_at":"2026-08-08T04:44:34.033813Z","submitted_at":"2026-06-26T20:36:26Z","title":"Embodiment Meets Environment: Toward Context-Aware, Safe Physical Caregiving Robots","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-30T00:56:51.666226Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2606.28592"},"observation_digest":"sha256:38efbd25d9389d9590f75b04f5618257d45b0ecaa553a6be43659eed5235d1f0","observation_id":"c6f0ebdf-5b56-4461-809a-c50d35f44a99","resolution":{"observed_at":"2026-07-01T15:55:49.243148Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-11T16:26:39.443485Z","title":"arXiv preprint arXiv:2306.03514 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04612","last_updated":"2026-07-06T02:37:22Z","snapshot_observed_at":"2026-08-06T16:33:39.351941Z","submitted_at":"2026-07-06T02:37:22Z","title":"StructuredEdit: Constraint-Aware Graphic Design Editing via Differentiable Parameter Propagation","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-07-11T16:26:39.443485Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2607.04612"},"observation_digest":"sha256:8e845f858f79a667aa2e12451f1b42ed18c71ed52065bdd7f48c8047b98cfa5c","observation_id":"dc268afb-10ac-4d4c-8bcc-ab301db38d8a","resolution":{"observed_at":"2026-07-11T16:26:39.443485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-01T22:34:46.291494Z","title":"arXiv preprint arXiv:2306.03514 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15711","last_updated":"2026-07-17T07:49:05Z","snapshot_observed_at":"2026-08-09T11:27:36.259203Z","submitted_at":"2026-07-17T07:49:05Z","title":"Efficient Difficulty-Aware Dynamic Routing for Diffusion-Based Real-World Image Super-Resolution","version":1},"reference_index":141,"source":"arxiv_source","source_observed_at":"2026-08-01T22:34:46.291494Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2607.15711"},"observation_digest":"sha256:499f4d7b5900a1d929ff816cdd3eea86401bd59c2a8d7ef8a2f3d154bd16e73c","observation_id":"bc185dca-cc97-409c-be44-371d52aaa6eb","resolution":{"observed_at":"2026-08-01T22:34:46.291494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2306.03514/citation-record","integrity":"/paper/2306.03514/integrity","json":"/paper/2306.03514/citation-record.json","paper":"/paper/2306.03514"},"outbound":[],"paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-10T03:01:07.020726Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 21 inbound Pith citation observations for arXiv:2306.03514."}