{"as_of":"2026-08-10T13:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7c7fda0752c0aa77aca5a1c7e86386c422e1b79fe6fa118dde283fed9000c62c","coverage":[{"denominator":108,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:22:15.365213Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.11102/citation-record","integrity":"/paper/2507.11102/integrity","json":"/paper/2507.11102/citation-record.json","paper":"/paper/2507.11102"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T17:22:08.397914Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.397914Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:637c5ea99c44a1cfc26bc5af76153f2ae898bf2e5c35df75d223d662808e3304","observation_id":"1e575dc4-462e-4c0d-8792-da60801519e8","resolution":{"observed_at":"2026-08-06T17:22:08.397914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.439737Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.439737Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:6f72200eaf6a0d820a2631fc6aa23d5061ff36ddc88677fd759b297e11d931eb","observation_id":"f21a9d29-8ac7-43af-8b26-d54d60991ba3","resolution":{"observed_at":"2026-08-06T17:22:08.439737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.500514Z","title":"2d human pose estimation: New benchmark and state of the art analysis","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.500514Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:3b6297e30e819e425d8bcb75ed0d7a895e494d94b704a685f33df9727f823959","observation_id":"7207cbbb-8cd1-42d0-b7aa-1c872101389e","resolution":{"observed_at":"2026-08-06T17:22:08.500514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-06T17:22:08.549301Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.549301Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:d56c9ce67034cfc7354aadf7e4e0d2e1589b3b224434d60c563e5ecaf6415153","observation_id":"9af81807-c9b1-483e-a1f1-5bd556b02e07","resolution":{"observed_at":"2026-08-06T17:22:08.549301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.604576Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.604576Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:ff447b7d61d6a3e1034062b525ca53ddebb5eb647799c51ef955fed67e5c43c8","observation_id":"52b6fe1e-6910-40d6-9465-debb48ae05f9","resolution":{"observed_at":"2026-08-06T17:22:08.604576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.657396Z","title":"Cross-domain adaptation for animal pose estimation","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.657396Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:3d2fa6468b0dfe3852fb8f5775639d4dc8745ca0098240e62117e580e7152308","observation_id":"3d5dcce1-3c34-4caf-bfc3-bc718347085b","resolution":{"observed_at":"2026-08-06T17:22:08.657396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15195","last_updated":"2023-07-03T16:08:00Z","snapshot_observed_at":"2026-07-06T15:47:07.545213Z","submitted_at":"2023-06-27T04:31:52Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15195","snapshot_observed_at":"2026-08-06T17:22:08.708494Z","title":"Shikra: Unleashing multimodal llm's referential dialogue magic","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.708494Z"},"links":{"cited_paper":"/paper/2306.15195","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:dd0ccccfbe8dec770b0c94582642115dc5c2357262c729c97b17503894d779a8","observation_id":"722e2f4f-6d8a-4e55-bd06-43be25a962dc","resolution":{"observed_at":"2026-08-06T17:22:08.708494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20340","last_updated":"2024-05-30T17:59:50Z","snapshot_observed_at":"2026-08-09T07:42:51.616278Z","submitted_at":"2024-05-30T17:59:50Z","title":"MotionLLM: Understanding Human Behaviors from Human Motions and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20340","snapshot_observed_at":"2026-08-06T17:22:08.784420Z","title":"Motionllm: Understanding human behaviors from human motions and videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.784420Z"},"links":{"cited_paper":"/paper/2405.20340","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:441b376efbea5de2dfee5c075ceb0da6b95dea14d0f4f85f7352f410b3e376a7","observation_id":"41e14b63-2c7c-442a-a5df-bdb37a9e620a","resolution":{"observed_at":"2026-08-06T17:22:08.784420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.873244Z","title":"Cascaded pyramid network for multi-person pose estimation","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.873244Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:746737d5c6922251929232d1c91a7b45e629b7d4a17338dbc624ea420e06847a","observation_id":"df60423c-1372-4f5a-a04d-352ff4cf9608","resolution":{"observed_at":"2026-08-06T17:22:08.873244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.937648Z","title":"Higherhrnet: Scale-aware representation learning for bottom-up human pose estimation","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.937648Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:8d928ec299ed04f017862e0585091a766aba4122f9adb891ced82300681c2f67","observation_id":"ad8c833b-17f8-450f-87cb-5f33596beff0","resolution":{"observed_at":"2026-08-06T17:22:08.937648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.022376Z","title":"Palm: Scaling language modeling with pathways","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.022376Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:b325276da65ac72f3ddfe49f33eab09062f031b3ccb6671451c183602843d398","observation_id":"adab5577-926b-4db0-a236-928264f079de","resolution":{"observed_at":"2026-08-06T17:22:09.022376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.091426Z","title":"Instructblip: Towards general-purpose vision-language models with instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.091426Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:e2e936231af59a7cefb027a16579f191f1807c70362ca8c1ac6e1e1945b42c3b","observation_id":"81eae217-1435-45f6-aa0d-867eeb2773ad","resolution":{"observed_at":"2026-08-06T17:22:09.091426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.153335Z","title":"Model-agnostic meta-learning for fast adaptation of deep networks","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.153335Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:36c4cdac84907b440b356a7ce51c5c409e6e49b37bbc27c0ca7d1cc202950239","observation_id":"c30815ce-5e4e-4f70-85a7-6f2e35045019","resolution":{"observed_at":"2026-08-06T17:22:09.153335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.221988Z","title":"Deepfashion2: A versatile benchmark for detection, pose estimation, segmentation and re-identification of clothing images","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.221988Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:f7ed27da204b15262d345b02a9ec685d44925f3861f9f84fb4a8fe45016a39ee","observation_id":"4273fe13-5bf9-4cbb-9c21-187d885d7cef","resolution":{"observed_at":"2026-08-06T17:22:09.221988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.03168","last_updated":"2025-02-28T14:39:17Z","snapshot_observed_at":"2026-08-08T10:55:01.205999Z","submitted_at":"2024-07-03T14:41:39Z","title":"LivePortrait: Efficient Portrait Animation with Stitching and Retargeting Control","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.03168","snapshot_observed_at":"2026-08-06T17:22:09.273146Z","title":"Liveportrait: Efficient portrait animation with stitching and retargeting control","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.273146Z"},"links":{"cited_paper":"/paper/2407.03168","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:04e0d1061139fb65ddfb758f9dd4dd2d43f95f5fa9910aff04899f18c64b71ab","observation_id":"63dbaaf5-b091-4f5f-912c-d04fd5006345","resolution":{"observed_at":"2026-08-06T17:22:09.273146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-07T07:43:16.294957Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-06T17:22:09.347197Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.347197Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:d9bcdbc41aebfe08adae382b91ced6e905f5065026b7b68ec216693cfa76e4e3","observation_id":"66112164-54e7-471d-afcc-d52f7409b51f","resolution":{"observed_at":"2026-08-06T17:22:09.347197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-06T17:22:09.421311Z","title":"Mistral 7b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.421311Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:c4d8c6d4e4387cd76f5e9b44a072d05dba0ae5beedb1c984ecdad70df90c5646","observation_id":"e20c208d-7fd7-46dc-86ac-a750eb9ee0c6","resolution":{"observed_at":"2026-08-06T17:22:09.421311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.503131Z","title":"Multi-person articulated tracking with spatial and temporal embeddings","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.503131Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:a7029a21df4c6f50eeac3eae5a9b048161ae7e750c84b7c423c69a0342878dac","observation_id":"edc05f21-d471-49ec-9075-97a12da249c4","resolution":{"observed_at":"2026-08-06T17:22:09.503131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.546326Z","title":"Differentiable hierarchical graph grouping for multi-person pose estimation","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.546326Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:772633800fb1d6072b13b2fd6007b40baecbcc2f67f80f6221c979b792e7e392","observation_id":"98051cca-7b82-4cd6-85fa-ce489709bbf6","resolution":{"observed_at":"2026-08-06T17:22:09.546326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.607010Z","title":"Whole-body human pose estimation in the wild","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.607010Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:3e426f54505b9858b7f2e6d90161c0cc17a7a8fd6840626d03756e2b2e025b13","observation_id":"28d9a720-b559-4054-85c2-507a171dcc03","resolution":{"observed_at":"2026-08-06T17:22:09.607010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.665516Z","title":"Human-art: A versatile human-centric dataset bridging natural and artificial scenes","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.665516Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:8d51b04239a6c301148992b0cb6ac5dfd04ba0378bdacbf64c449270c8152edb","observation_id":"acb0fec6-1929-4787-8c39-1d07c9d5ca6f","resolution":{"observed_at":"2026-08-06T17:22:09.665516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.732831Z","title":"Humansd: A native skeleton-guided diffusion model for human image generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.732831Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:e94c3ea432df54bc7f29aea4ef890b6fe809d6d1e830d11d5b74ebb146f8f473","observation_id":"aa15c8a4-e594-438d-89e3-6b7be211d618","resolution":{"observed_at":"2026-08-06T17:22:09.732831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.795152Z","title":"Animalweb: A large-scale hierarchical dataset of annotated animal faces","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.795152Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:6bea37617609977794812f0b0a176c720d41afab167b9bfbf5b58d3a7d4d21af","observation_id":"19cfd8aa-c4c9-4adf-aa5a-6c3e1a57a06f","resolution":{"observed_at":"2026-08-06T17:22:09.795152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.853259Z","title":"Segment anything","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.853259Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:bbdf721dba3d79e1d5418a850351d4422c18bfb800a93856b84b680559275320","observation_id":"965e575e-c9ea-4af7-b542-be3f8a7b3ca7","resolution":{"observed_at":"2026-08-06T17:22:09.853259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.915013Z","title":"in the wild","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.915013Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:41857093b116e2dc8dfb62760cb5140af8a7505405a2396589885c20bc7d621e","observation_id":"3e53591d-314e-4c55-a300-da9b080d313c","resolution":{"observed_at":"2026-08-06T17:22:09.915013Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.00692","last_updated":"2024-05-01T05:10:13Z","snapshot_observed_at":"2026-08-06T07:43:56.889679Z","submitted_at":"2023-08-01T17:50:17Z","title":"LISA: Reasoning Segmentation via Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.00692","snapshot_observed_at":"2026-08-06T17:22:09.963924Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.963924Z"},"links":{"cited_paper":"/paper/2308.00692","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:4472274267442bdad4fc67cb4660308c3521f38f8bcac9c8795168c57ac63ac6","observation_id":"096df928-9178-48e5-8a28-324a82b4db17","resolution":{"observed_at":"2026-08-06T17:22:09.963924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.02246","last_updated":"2024-05-03T17:00:00Z","snapshot_observed_at":"2026-08-08T17:33:57.727194Z","submitted_at":"2024-05-03T17:00:00Z","title":"What matters when building vision-language models?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.02246","snapshot_observed_at":"2026-08-06T17:22:10.040901Z","title":"What matters when building vision-language models? arXiv preprint arXiv:2405.02246, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.040901Z"},"links":{"cited_paper":"/paper/2405.02246","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:304525849a0548493ff1b20bdaadf69eaeb9f5ae93c6f9d26e47315c58125e79","observation_id":"41980a60-03c0-424b-914f-2f2352626e52","resolution":{"observed_at":"2026-08-06T17:22:10.040901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05425","last_updated":"2023-06-08T17:59:56Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"MIMIC-IT: Multi-Modal In-Context Instruction Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05425","snapshot_observed_at":"2026-08-06T17:22:10.082980Z","title":"Mimic-it: Multi-modal in-context instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.082980Z"},"links":{"cited_paper":"/paper/2306.05425","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:eb8a1e0fd02735df5f5f9b1bb041015556b44939438066aed6997e5004060f44","observation_id":"e5a426ab-7b91-4d0e-8765-35979addef97","resolution":{"observed_at":"2026-08-06T17:22:10.082980Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.140423Z","title":"Crowdpose: Efficient crowded scenes pose estimation and a new benchmark","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.140423Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:19e886a77144aa02532d21bb9e7bb2ec2b8281f70ce701a425262e289bd450ee","observation_id":"0831afbf-fc6c-4d6d-a7f9-89bacda92b36","resolution":{"observed_at":"2026-08-06T17:22:10.140423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.212644Z","title":"Human pose regression with residual log-likelihood estimation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.212644Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:f47a93c1ce774dca2575b7b861f7b3e257adfff40ba677ec63298023623295a7","observation_id":"35498739-a94d-48f3-866a-72b9c207bf4d","resolution":{"observed_at":"2026-08-06T17:22:10.212644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.277017Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.277017Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:1bf43b6984f51d7cb516f94e16ad746494c1787bd3500c2e6e9fdb8247433142","observation_id":"7becd0fc-4cd5-4589-a604-332897af47e1","resolution":{"observed_at":"2026-08-06T17:22:10.277017Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.03516","last_updated":"2021-08-13T15:25:09Z","snapshot_observed_at":"2026-08-07T09:29:02.943596Z","submitted_at":"2021-04-08T05:12:38Z","title":"TokenPose: Learning Keypoint Tokens for Human Pose Estimation","version":3},"cited_work":{"arxiv_id":"2104.03516","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.03516","snapshot_observed_at":"2026-08-06T17:22:16.445473Z","title":"TokenPose: Learning Keypoint Tokens for Human Pose Estimation","venue":"cs.CV","work_id":"10845080-1946-4cbd-add1-e1dcb01ccbf6","year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.347059Z"},"links":{"cited_paper":"/paper/2104.03516","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:539e2c4240e76893a24a1ad55c2e5e9d590e8c4c51d4be818874844e6ec09df8","observation_id":"aca1c1be-834d-41ae-aeb5-8c43a9a3e5ca","resolution":{"observed_at":"2026-08-06T17:22:16.549635Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.392295Z","title":"Simcc: A simple coordinate classification perspective for human pose estimation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.392295Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:4c35d61e08b7a83d97d7ae5ca3e64856c71686a268bf19cfddd3a9894900603b","observation_id":"891c5c39-6db8-43bc-8402-b29843605d37","resolution":{"observed_at":"2026-08-06T17:22:10.392295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.18814","last_updated":"2024-03-27T17:59:04Z","snapshot_observed_at":"2026-07-31T05:41:28.385099Z","submitted_at":"2024-03-27T17:59:04Z","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.18814","snapshot_observed_at":"2026-08-06T17:22:10.455266Z","title":"Mini-gemini: Mining the potential of multi-modality vision language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.455266Z"},"links":{"cited_paper":"/paper/2403.18814","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:60150bcd9042e7f598f2710d4c6d407ecffcf29abe6fc7a6d32d6b96a4b2f944","observation_id":"6069bdca-5dab-4ef4-a61d-a758b62f7f20","resolution":{"observed_at":"2026-08-06T17:22:10.455266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.07533","last_updated":"2024-05-16T21:21:30Z","snapshot_observed_at":"2026-08-10T13:01:53.292743Z","submitted_at":"2023-12-12T18:58:18Z","title":"VILA: On Pre-training for Visual Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.07533","snapshot_observed_at":"2026-08-06T17:22:10.522038Z","title":"Vila: On pre-training for visual language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.522038Z"},"links":{"cited_paper":"/paper/2312.07533","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:984659dab2df154b75007b6a66d9fd705ed663111569168769d2c9dd51673fb4","observation_id":"08768bc4-df80-43f8-a3c7-0cdc7de55359","resolution":{"observed_at":"2026-08-06T17:22:10.522038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.573970Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.573970Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:b88908a095accfa689a09dbee22e4ab604f5feb89ee824bcc9e90a2274beda86","observation_id":"eefea3c9-1154-4999-9ecf-ae97a6a6659e","resolution":{"observed_at":"2026-08-06T17:22:10.573970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.659686Z","title":"Improved baselines with visual instruction tuning, 2023 a","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.659686Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:ca64b046b3970f791e72ac365b74342ca5d28513bc662cb6359cd443b2be4fc5","observation_id":"29dfe3b6-5bb1-4e61-a502-9567b7b7ff88","resolution":{"observed_at":"2026-08-06T17:22:10.659686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.712260Z","title":"Visual instruction tuning, 2023 b","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.712260Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:683937b847e5c78e18ca3fb2030703e30d76bb745a37ccd5c3bc66bc3bf860f0","observation_id":"ae8ffabf-0df9-44f0-8646-2f4000136edc","resolution":{"observed_at":"2026-08-06T17:22:10.712260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.756326Z","title":"Llava-next: Improved reasoning, ocr, and world knowledge, January 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.756326Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:441b17a9f651063cc68c259f2bc07022a083ff204c67b680a4fe6a68227f4c9d","observation_id":"c9756c63-7999-4256-b83d-8bb282882a66","resolution":{"observed_at":"2026-08-06T17:22:10.756326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.818362Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.818362Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:701baecda81e02394daaea07c4a226425a92383ad3351397d3b2d7946e76b881","observation_id":"f30433d3-3016-47b7-90e4-eea774095811","resolution":{"observed_at":"2026-08-06T17:22:10.818362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.875282Z","title":"A convnet for the 2020s","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.875282Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:2bdba44ca591d1fbb9d021df0fa57a1b105d860be6d6ab9990703358b6b34367","observation_id":"daa8169b-1f48-4de4-ae48-fff91bc58e1a","resolution":{"observed_at":"2026-08-06T17:22:10.875282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.920288Z","title":"Deepseek-vl: Towards real-world vision-language understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.920288Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:5bc460f0bc4fb0f87726df543571b8a2d3edc89ad6aee7adc4caa5758b7bc7e6","observation_id":"939d3dcc-b1d7-4e5e-b825-3378cd0e8548","resolution":{"observed_at":"2026-08-06T17:22:10.920288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12978","last_updated":"2023-10-19T17:59:46Z","snapshot_observed_at":"2026-08-03T23:57:00.735076Z","submitted_at":"2023-10-19T17:59:46Z","title":"HumanTOMATO: Text-aligned Whole-body Motion Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12978","snapshot_observed_at":"2026-08-06T17:22:10.990900Z","title":"Humantomato: Text-aligned whole-body motion generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.990900Z"},"links":{"cited_paper":"/paper/2310.12978","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:6ff113fea0bd795b0afc7b054657f7cb329960e1eae89ad0a8b610c5fb54a7c4","observation_id":"6e3305b5-61ea-4571-a060-fea05536bf93","resolution":{"observed_at":"2026-08-06T17:22:10.990900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.619568Z","title":"From keypoints to object landmarks via self-training correspondence: A novel approach to unsupervised landmark discovery","venue":null,"work_id":"e85f8db6-b6c8-4adc-a004-f1513937600a","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.997048Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:af2a8eb58682cfc6adcc5ea087c8e60154b216084eb5d7851f4ba8677453e062","observation_id":"b6d07928-854b-4ce8-931b-60c564c9f891","resolution":{"observed_at":"2026-08-06T17:22:22.693855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.09611","last_updated":"2024-04-18T18:51:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-14T17:51:32Z","title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.09611","snapshot_observed_at":"2026-08-06T17:22:11.164532Z","title":"Mm1: Methods, analysis & insights from multimodal llm pre-training","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.164532Z"},"links":{"cited_paper":"/paper/2403.09611","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:0e4ac073dd587e333a99e7cd66b20ce756561e4956b175307a510908d0d4a911","observation_id":"a5b53bb2-96bb-493c-b041-ddd1971f8bb5","resolution":{"observed_at":"2026-08-06T17:22:11.164532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08295","last_updated":"2024-04-16T12:52:47Z","snapshot_observed_at":"2026-08-03T03:29:01.959523Z","submitted_at":"2024-03-13T06:59:16Z","title":"Gemma: Open Models Based on Gemini Research and Technology","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08295","snapshot_observed_at":"2026-08-06T17:22:11.281720Z","title":"Gemma: Open models based on gemini research and technology","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.281720Z"},"links":{"cited_paper":"/paper/2403.08295","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:3eb0547fb18f0a9354a793a6bd18403f040401b90e461a96a57ed58a84eb82b3","observation_id":"37e83a97-d2bc-4ebe-8f31-31181833f824","resolution":{"observed_at":"2026-08-06T17:22:11.281720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.450534Z","title":"Interhand2.6m: A dataset and baseline for 3d interacting hand pose estimation from a single rgb image","venue":null,"work_id":"4327c0ac-7f89-4614-ae1c-20fa76bb633a","year":2020},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.337039Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:3bf364bc7aa28f554c01c63be877b512b3b1bdaa6e542007be43f7a922924aa8","observation_id":"f3bb7e68-4f45-46ec-bb30-0a496f696331","resolution":{"observed_at":"2026-08-06T17:22:22.522833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.00216","last_updated":"2019-10-03T04:53:10Z","snapshot_observed_at":"2026-07-06T08:25:58.847575Z","submitted_at":"2019-10-01T06:21:50Z","title":"Revisiting Fine-tuning for Few-shot Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.00216","snapshot_observed_at":"2026-08-06T17:22:11.496851Z","title":"Revisiting fine-tuning for few-shot learning","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.496851Z"},"links":{"cited_paper":"/paper/1910.00216","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:cd15dc7cdef1825d6f932d631d6507eb3b86f534603d601489a7420027a8c27c","observation_id":"9dba3041-d485-46d1-ab8e-c9ffaa0219d1","resolution":{"observed_at":"2026-08-06T17:22:11.496851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.355781Z","title":"Stacked hourglass networks for human pose estimation","venue":null,"work_id":"93cf2665-4d6c-499a-8b98-bd67ef78ad68","year":2016},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.582583Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:9f54e0dfedfcedd78b14f9869ed64ee014b27ff60a2d184a21f411d2decea6c8","observation_id":"dc1ef1a2-29a6-48dd-92fa-dec6eceb1cdd","resolution":{"observed_at":"2026-08-06T17:22:22.402703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.248725Z","title":"Animal kingdom: A large and diverse dataset for animal behavior understanding","venue":null,"work_id":"63506b3c-6e3a-46bf-88f8-9cd0a39a9b1f","year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.692051Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:77707c4e68ab4a2df6a17575c8fb8ed77360455882a843978699bd1c7be46833","observation_id":"f6cdb89f-0137-4137-9b5e-dcc93f6750bb","resolution":{"observed_at":"2026-08-06T17:22:22.305728Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.167882Z","title":"Single-stage multi-person pose machines","venue":null,"work_id":"33ad63d2-333e-4afe-acd5-8b1424c0b82a","year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.803010Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:695abc060a7844b66c4117f9c4f75cd23dc247c4c64d175e8f923ae4a8ebce96","observation_id":"9bacbe1e-dc12-4098-b34a-28ed5de5a620","resolution":{"observed_at":"2026-08-06T17:22:22.202969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-08-09T18:02:17.307812Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-06T17:22:11.891931Z","title":"Dinov2: Learning robust visual features without supervision","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.891931Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:733ee22456e22c47c9de5a1f3a05ea5cc02cade15db4a4576479be5a1cfe7bbf","observation_id":"cec821b3-8e61-44ef-94e5-0bc0296fc372","resolution":{"observed_at":"2026-08-06T17:22:11.891931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03277","last_updated":"2023-04-06T17:58:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-06T17:58:09Z","title":"Instruction Tuning with GPT-4","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03277","snapshot_observed_at":"2026-08-06T17:22:12.010239Z","title":"Instruction tuning with gpt-4","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.010239Z"},"links":{"cited_paper":"/paper/2304.03277","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:2da0a251664ecbaa9937f4f2a163c5e172ee69eb24c9faae43b97e4336c7b1c5","observation_id":"d5bacdeb-da28-496e-948a-7b8e26a9a10b","resolution":{"observed_at":"2026-08-06T17:22:12.010239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14824","last_updated":"2023-07-13T05:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-26T16:32:47Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14824","snapshot_observed_at":"2026-08-06T17:22:12.091020Z","title":"Kosmos-2: Grounding multimodal large language models to the world","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.091020Z"},"links":{"cited_paper":"/paper/2306.14824","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:a5d81197b7b144eec9a87b4f4c8359bd777e5ec38acbde07059cfab4a85b4bc7","observation_id":"2f35e897-2827-487e-a2d1-71341e8baec7","resolution":{"observed_at":"2026-08-06T17:22:12.091020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14167","last_updated":"2023-05-24T02:51:37Z","snapshot_observed_at":"2026-08-06T19:11:00.161773Z","submitted_at":"2023-05-23T15:37:28Z","title":"DetGPT: Detect What You Need via Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14167","snapshot_observed_at":"2026-08-06T17:22:12.156699Z","title":"Detgpt: Detect what you need via reasoning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.156699Z"},"links":{"cited_paper":"/paper/2305.14167","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:a4583c7a1ea13794aeb807c2f1994d21938c6320e663c8a9f8f9882b4d286f75","observation_id":"f0717450-6257-4e36-8e3a-1b0e092292d2","resolution":{"observed_at":"2026-08-06T17:22:12.156699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06612","last_updated":"2023-11-11T16:59:20Z","snapshot_observed_at":"2026-07-06T16:46:06.801011Z","submitted_at":"2023-11-11T16:59:20Z","title":"PerceptionGPT: Effectively Fusing Visual Perception into LLM","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.06612","snapshot_observed_at":"2026-08-06T17:22:12.222475Z","title":"Perceptiongpt: Effectively fusing visual perception into llm","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.222475Z"},"links":{"cited_paper":"/paper/2311.06612","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:a9b50bf8dcae3cffce5f973e6cb178aa81c81a2896389162705c6e12592af7d3","observation_id":"ac7378b4-d987-40b7-aad6-4158d99f5e3e","resolution":{"observed_at":"2026-08-06T17:22:12.222475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.093976Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"88d5a625-2a02-4551-8ead-088ac2e9d03f","year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.301281Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:2d7ba5a0c37e6091439faf02cde21d6d5f2a6d2f1dc15dd1a3840406807f9fb6","observation_id":"00f6b4e1-e771-444b-8855-7ade893bbde3","resolution":{"observed_at":"2026-08-06T17:22:22.125639Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.001906Z","title":"Carfusion: Combining point tracking and part detection for dynamic 3d reconstruction of vehicles","venue":null,"work_id":"264d61a4-1416-448d-80f7-d64fc24fafb2","year":2018},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.360012Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:9210af063fc0d8fa4842d16960fa87f0c7a7fb4d0c84c30d89681b774082ddea","observation_id":"ab397d07-183e-4853-b753-63ea015c68a6","resolution":{"observed_at":"2026-08-06T17:22:22.053927Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.800841Z","title":"Zafeiriou, and M","venue":null,"work_id":"9477b7fe-c449-4713-b755-bc6f990f3dad","year":2016},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.443453Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:f88d8589cc848f49387444e1bf210b34205b9bd7c8bd9b7b7cfad15a7db8af33","observation_id":"c809e9d7-db83-40d6-82ee-37b313a06f21","resolution":{"observed_at":"2026-08-06T17:22:21.902725Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.632737Z","title":"Matching is not enough: A two-stage framework for category-agnostic pose estimation","venue":null,"work_id":"5eddc45c-a2db-41e0-aa0b-e4cc1104673c","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.518555Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:570a7de1822c6092c1181c1f67535caae9bd836bb9f3681e75283ee6ac20ab29","observation_id":"193d7a0f-d084-44a4-b50e-334e49b1ad21","resolution":{"observed_at":"2026-08-06T17:22:21.697796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.516967Z","title":"Prototypical networks for few-shot learning","venue":null,"work_id":"cf851491-4f97-4eec-9f78-095d56483edf","year":2017},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.586243Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:fd8b85612cf4c649f4c8796a5620acc7a941e80534295155c698e7b164ad447d","observation_id":"0044afed-fb6e-4f5d-92b4-7bd38c1191e3","resolution":{"observed_at":"2026-08-06T17:22:21.562706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.408978Z","title":"Self-supervised keypoint discovery in behavioral videos","venue":null,"work_id":"0f50cdde-484b-4bdd-9996-1446d5386666","year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.645354Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:e570917788c0e0419fb2848626725513a38083e2ce232ffd151072d3e8a1c57a","observation_id":"d707b8cd-7738-4732-931a-b313d78d12ae","resolution":{"observed_at":"2026-08-06T17:22:21.473712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.289630Z","title":"Deep high-resolution representation learning for human pose estimation","venue":null,"work_id":"0b39eb2b-d11f-4e1a-a4ff-002e9de7f4fc","year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.715825Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:86469e307193f261e850ec7ebc7cfcc217eedab298ef690438dd12d1bed033ae","observation_id":"4a5fb059-5cab-44c9-8db4-85a02e3febdd","resolution":{"observed_at":"2026-08-06T17:22:21.322446Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.115262Z","title":"Compositional human pose regression","venue":null,"work_id":"2f0178d7-4b30-4d7c-8021-e8f60a610cb0","year":2017},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.772097Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:ad99071c4fbf1b59ca7439280010602ac95fe6c64aef827ba50a23609f45cf18","observation_id":"9d19d238-f1ea-4518-a951-d71afbf21c20","resolution":{"observed_at":"2026-08-06T17:22:21.212839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:20.951726Z","title":"Deeppose: Human pose estimation via deep neural networks","venue":null,"work_id":"e1ecbc3f-0939-4a42-8c33-c33c12e01e65","year":2014},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.848054Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:fa676a460c99be4886a6d8ce10ace28d40703895af2a598b56ec1d017d94892a","observation_id":"020a2b09-2843-435d-96e9-b60b71384085","resolution":{"observed_at":"2026-08-06T17:22:20.996690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-06T17:22:12.913739Z","title":"Llama: Open and efficient foundation language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.913739Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:b08272682c7248b4dc49237153b05dc7bf57bb77f76a2edfa14f71ad29a0f2fa","observation_id":"11fd6029-7466-4628-9867-b98b512596c8","resolution":{"observed_at":"2026-08-06T17:22:12.913739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-06T17:22:13.017689Z","title":"Llama 2: Open foundation and fine-tuned chat models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.017689Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:ff363db2ee4fe877e27880bcfd28046d5dbe7aec1cee55a2bf6a8a7521d369d0","observation_id":"ab97ce07-9584-4586-8107-d90835be4ba1","resolution":{"observed_at":"2026-08-06T17:22:13.017689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:20.835637Z","title":"Attention is all you need","venue":null,"work_id":"71e54be0-e7ce-47cf-8a1f-0381818ede5a","year":2017},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.097863Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:01f067976210069c71538294ffec22b44cca26f5736d0f27d55c572c7dcad968","observation_id":"c6e17968-63d8-4058-99ed-d85d64764396","resolution":{"observed_at":"2026-08-06T17:22:20.889936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:20.654754Z","title":"Locllm: Exploiting generalizable human keypoint localization via large language model","venue":null,"work_id":"60a84d01-b145-4248-84ad-32792c3599b8","year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.162173Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:e094a920ebb5560bb9fb10c928d620c15d3c2e0c54e19dd5a6f01c917f027561","observation_id":"1682b3c2-1b56-4d90-954f-4680153e4297","resolution":{"observed_at":"2026-08-06T17:22:20.760112Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:20.384593Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"852f2dd0-b846-453b-b670-dd71a634e705","year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.224523Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:229365d5e038762778bf33c2e0a88b5430176a717ba547ddbd3127303fe4f7af","observation_id":"1093694e-ff0d-42cc-b21b-0d6b8cc879ef","resolution":{"observed_at":"2026-08-06T17:22:20.495122Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:20.110723Z","title":"Convolutional pose machines","venue":null,"work_id":"07f6d653-32a9-4e81-973b-8fdef9f80301","year":2016},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.325920Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:d74fd0fd544ca5e4ecfcff9208729ddd9c42541921568839f3565f55d9a55f68","observation_id":"54aec589-a372-47f7-a2a5-8c701f4d0232","resolution":{"observed_at":"2026-08-06T17:22:20.249219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-08-06T17:22:13.366819Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model for hundreds of vision-language tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.366819Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:fe0999f72f600f82148147cf99a10a41e4f830efeea72189da2aabee05f8882f","observation_id":"d43a7054-ee52-4ca0-8c0d-5cc729736004","resolution":{"observed_at":"2026-08-06T17:22:13.366819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05821","last_updated":"2025-04-11T14:21:24Z","snapshot_observed_at":"2026-07-06T18:27:50.189771Z","submitted_at":"2024-06-09T15:14:26Z","title":"F-LMM: Grounding Frozen Large Multimodal Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05821","snapshot_observed_at":"2026-08-06T17:22:13.424856Z","title":"F-lmm: Grounding frozen large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.424856Z"},"links":{"cited_paper":"/paper/2406.05821","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:36890fc04b43571236cee63e2706e3b8e450cb00d58c53f3e46e60473b35a859","observation_id":"8029c7b6-d77d-49c4-a4dd-6d99ac6b047d","resolution":{"observed_at":"2026-08-06T17:22:13.424856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:19.924872Z","title":"Look at boundary: A boundary-aware face alignment algorithm","venue":null,"work_id":"88dff078-3c67-4abf-82af-7747e1015351","year":2018},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.517966Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:c0384151f652a7c06277d4d612d71d066dc92b7df6a75dce69c1d97bf2b69221","observation_id":"daa5e2e4-0e2d-460b-8b5b-bb93ef646c21","resolution":{"observed_at":"2026-08-06T17:22:19.991106Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:19.720960Z","title":"Simple baselines for human pose estimation and tracking","venue":null,"work_id":"1bed8c8e-b324-4283-bb00-cd25013fa5e1","year":2018},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.609050Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:bfc8342d136132a9c40dd2225af9e09b098e714eaed634d9dfd8ce72cf88350b","observation_id":"a8625e9d-9e14-4964-a116-79d40c55de5d","resolution":{"observed_at":"2026-08-06T17:22:19.816933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:19.503875Z","title":"Pixel-aligned language model","venue":null,"work_id":"70d343c1-43e0-4f3e-9be8-f115c844ad2e","year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.656756Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:4214d2b3c11a626a70fc54e1c127c75ad9129623b751445ac1460e257d625dec","observation_id":"4af29d8b-5f40-4ccb-91ba-6ff19c44019e","resolution":{"observed_at":"2026-08-06T17:22:19.613550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:19.183899Z","title":"Vipnas: Efficient video pose estimation via neural architecture search","venue":null,"work_id":"8f1bae23-34b1-4b25-a167-93c951e5be5c","year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.687889Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:dda0409cd7865cba45443a5caeb6437b06b27498bd9d63b7a6094f7d7d24ee5d","observation_id":"1b189f69-544b-4dbb-b977-e3fc681299bf","resolution":{"observed_at":"2026-08-06T17:22:19.315519Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:19.014590Z","title":"Pose for everything: Towards category-agnostic pose estimation","venue":null,"work_id":"666799f4-2ba2-4f37-895a-f2f9e16b6c6d","year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.770149Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:0472fb1a1797127d64aa8f6c0680376486085d48851c0f990c1474e33035c6d8","observation_id":"f51d56c7-31ae-4ef5-986b-2b22187b2ddc","resolution":{"observed_at":"2026-08-06T17:22:19.077852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.858176Z","title":"Vitpose: Simple vision transformer baselines for human pose estimation","venue":null,"work_id":"612d2b96-3a85-43a7-9978-9ede8ce12966","year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.820520Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:4d290ea41713eb0d14160a304d39e7b938e154b914cef1f68c330ddb25c93639","observation_id":"d665a2f6-f302-474a-9f64-27e72b9dfe03","resolution":{"observed_at":"2026-08-06T17:22:18.920419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12252","last_updated":"2023-05-20T17:59:23Z","snapshot_observed_at":"2026-07-06T15:30:05.820261Z","submitted_at":"2023-05-20T17:59:23Z","title":"Boosting Human-Object Interaction Detection with Text-to-Image Diffusion Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12252","snapshot_observed_at":"2026-08-06T17:22:13.857000Z","title":"Boosting human-object interaction detection with text-to-image diffusion model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.857000Z"},"links":{"cited_paper":"/paper/2305.12252","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:5284c3796bf54125729e7082993daaed1003a4b94b59f82f148453c9d8e4da34","observation_id":"623d16ab-ef14-4aeb-8c7f-ac9bdd74423a","resolution":{"observed_at":"2026-08-06T17:22:13.857000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.729319Z","title":"Semantic human parsing via scalable semantic transfer over multiple label domains","venue":null,"work_id":"8ea9f98d-98cb-4bdc-8ff4-e34d34a53de7","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.935885Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:8728dcd197af43428027ae76c3e4148807d8e716eb3fb5b8a449204b6a79b67e","observation_id":"e48fbddf-4a94-4d53-982f-5612c5b30cd0","resolution":{"observed_at":"2026-08-06T17:22:18.797167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.607339Z","title":"Neural interactive keypoint detection","venue":null,"work_id":"57526ba3-090f-4571-9c5b-80fd970bd322","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.997849Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:0d33a285709ec79110da05165b48782387f6e9ca89db94a4f7e9cc4ec687a87d","observation_id":"d92d8528-f971-418f-befa-a03528242823","resolution":{"observed_at":"2026-08-06T17:22:18.660229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01593","last_updated":"2023-02-03T08:18:34Z","snapshot_observed_at":"2026-08-09T02:23:24.822886Z","submitted_at":"2023-02-03T08:18:34Z","title":"Explicit Box Detection Unifies End-to-End Multi-Person Pose Estimation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01593","snapshot_observed_at":"2026-08-06T17:22:14.093641Z","title":"Explicit box detection unifies end-to-end multi-person pose estimation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.093641Z"},"links":{"cited_paper":"/paper/2302.01593","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:e6eea46b77c431a90d09a3b8df78181a5b6c8db82726395d3fe6bfdd33f0ae0e","observation_id":"d6f0255e-3426-4a8b-b49c-dedc33e2547f","resolution":{"observed_at":"2026-08-06T17:22:14.093641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08530","last_updated":"2024-07-17T09:25:24Z","snapshot_observed_at":"2026-07-06T16:32:07.186600Z","submitted_at":"2023-10-12T17:22:58Z","title":"X-Pose: Detecting Any Keypoints","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08530","snapshot_observed_at":"2026-08-06T17:22:14.151635Z","title":"Unipose: Detecting any keypoints","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.151635Z"},"links":{"cited_paper":"/paper/2310.08530","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:48a4e78264493d0e0154009bd0da45f1c38f0fc1f4af91b27b089d24ecccd9a3","observation_id":"14404574-6dac-455e-868a-156d94e504ad","resolution":{"observed_at":"2026-08-06T17:22:14.151635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12435","last_updated":"2024-07-17T09:43:58Z","snapshot_observed_at":"2026-08-07T06:38:54.672004Z","submitted_at":"2024-07-17T09:43:58Z","title":"F-HOI: Toward Fine-grained Semantic-Aligned 3D Human-Object Interactions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.12435","snapshot_observed_at":"2026-08-06T17:22:14.216807Z","title":"F-hoi: Toward fine-grained semantic-aligned 3d human-object interactions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.216807Z"},"links":{"cited_paper":"/paper/2407.12435","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:9daa8c5019bd651af18f06b1fc2c0042286a355caa795bcd52827dcee38b9dac","observation_id":"fbdff743-f45d-4789-a08a-b3958541ca59","resolution":{"observed_at":"2026-08-06T17:22:14.216807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.452549Z","title":"Kptllm: Unveiling the power of large language model for keypoint comprehension","venue":null,"work_id":"c341664f-73e8-410b-bd01-e79cd0ae43a3","year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.279793Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:693da218a217e1bf8e24b85cadee81d1404c0b1e79927c7ae373d7cac330fc73","observation_id":"4ba0d7ec-2233-4007-aaac-5d034fe7b636","resolution":{"observed_at":"2026-08-06T17:22:18.516527Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.294942Z","title":"Ed-pose++: Enhanced explicit box detection for conventional and interactive multi-object keypoint detection","venue":null,"work_id":"503f4101-3fc5-4a40-8879-b056042bab7f","year":2025},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.319317Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:9bfc2fdeff61b36b2202660bce0c93fbb4fa1270c6eea0a124a7e0cf2706b1d5","observation_id":"99992b3d-816a-4194-87e7-bafe0445bb99","resolution":{"observed_at":"2026-08-06T17:22:18.361369Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.151068Z","title":"Apt-36k: A large-scale benchmark for animal pose estimation and tracking","venue":null,"work_id":"ec6d6efe-573d-40e2-a3df-9460a4501948","year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.386205Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:af16a2432675969a983d1276aadfe0ba78008f514d20f07b97256ea1a180939f","observation_id":"c17f125e-98d2-4054-8b8b-f13350b71b81","resolution":{"observed_at":"2026-08-06T17:22:18.199140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-06T17:22:14.478484Z","title":"mplug-owl: Modularization empowers large language models with multimodality","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.478484Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:11c1ddb197781344607c7b9ef6fb19684183733f074affb8e8d9558690843bcf","observation_id":"ec2be9e1-6a25-49e9-92ff-b38306b2aec1","resolution":{"observed_at":"2026-08-06T17:22:14.478484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07704","last_updated":"2023-10-11T17:55:15Z","snapshot_observed_at":"2026-07-06T16:31:25.350087Z","submitted_at":"2023-10-11T17:55:15Z","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07704","snapshot_observed_at":"2026-08-06T17:22:14.559435Z","title":"Ferret: Refer and ground anything anywhere at any granularity","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.559435Z"},"links":{"cited_paper":"/paper/2310.07704","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:bd10ac0ea8f816302bc108517559bfe46b0a5a4826462085d703a3407bf276fc","observation_id":"53a9ac90-37f5-488d-9b45-56ab876e9f8c","resolution":{"observed_at":"2026-08-06T17:22:14.559435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.12617","last_updated":"2021-11-01T05:36:12Z","snapshot_observed_at":"2026-07-06T11:42:16.189770Z","submitted_at":"2021-08-28T10:23:34Z","title":"AP-10K: A Benchmark for Animal Pose Estimation in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.12617","snapshot_observed_at":"2026-08-06T17:22:14.658414Z","title":"Ap-10k: A benchmark for animal pose estimation in the wild","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.658414Z"},"links":{"cited_paper":"/paper/2108.12617","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:b24cee0a753f62554330700fee3caa9033d4e1487ea95ccc2c57d8aeb551821b","observation_id":"4bc9b194-deaa-45f1-95cb-ef08bbd1047d","resolution":{"observed_at":"2026-08-06T17:22:14.658414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.09408","last_updated":"2021-11-07T14:39:41Z","snapshot_observed_at":"2026-07-06T11:59:08.372025Z","submitted_at":"2021-10-18T15:37:58Z","title":"HRFormer: High-Resolution Transformer for Dense Prediction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.09408","snapshot_observed_at":"2026-08-06T17:22:14.730911Z","title":"Hrformer: High-resolution transformer for dense prediction","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.730911Z"},"links":{"cited_paper":"/paper/2110.09408","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:2f8e1ae92025177ff2af90b7c75358014bf247cb2e1920550ac95fa7af6ff9bd","observation_id":"2d8289d9-32c5-4689-b2d0-f2de7b9836a2","resolution":{"observed_at":"2026-08-06T17:22:14.730911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18279","last_updated":"2024-08-12T07:14:00Z","snapshot_observed_at":"2026-08-10T04:55:17.331940Z","submitted_at":"2023-05-29T17:50:33Z","title":"Contextual Object Detection with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.18279","snapshot_observed_at":"2026-08-06T17:22:14.824478Z","title":"Contextual object detection with multimodal large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.824478Z"},"links":{"cited_paper":"/paper/2305.18279","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:c46463a42e115c55bb7838dec2898de1c022e5e62cc2ac1c5b61f132de5b1d3a","observation_id":"23390d0a-6e3d-4aa9-8eb2-366838c9140e","resolution":{"observed_at":"2026-08-06T17:22:14.824478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:17.976686Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"d9057dff-7b3a-4d5c-9eee-b5f72b73ff38","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.952673Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:7585b8ba2562c282aa087e70524423dcf1e49948c88f9ba7f1ce8a98a7c8c179","observation_id":"1378ab1f-f0eb-418a-bf65-923ae2222a24","resolution":{"observed_at":"2026-08-06T17:22:18.057396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:17.825494Z","title":"Open-vocabulary animal keypoint detection with semantic-feature matching","venue":null,"work_id":"0d8bbf32-a9b1-4dc1-9558-27b712a103a1","year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.020088Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:0ec1805c337e0a95ec119f78d567fc9f355ddb17ea62904b3a92e83e436ce5a4","observation_id":"f921b1a8-6eff-4729-a310-922fc4c6a742","resolution":{"observed_at":"2026-08-06T17:22:17.934887Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:17.575802Z","title":"Adding conditional control to text-to-image diffusion models","venue":null,"work_id":"53206848-b1aa-4748-b6da-dff0865631c7","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.070625Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:35d4f3fb959f018bfd6bec2310ed2d535f873eebb7141159e484a93b3f283537","observation_id":"4447a533-086c-48e3-a756-4810421bc257","resolution":{"observed_at":"2026-08-06T17:22:17.682866Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.03601","last_updated":"2025-06-12T00:15:18Z","snapshot_observed_at":"2026-08-07T18:00:59.339869Z","submitted_at":"2023-07-07T13:43:44Z","title":"GPT4RoI: Instruction Tuning Large Language Model on Region-of-Interest","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.03601","snapshot_observed_at":"2026-08-06T17:22:15.142692Z","title":"Gpt4roi: Instruction tuning large language model on region-of-interest","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.142692Z"},"links":{"cited_paper":"/paper/2307.03601","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:047b2328940adc7718c34db5e8e9f1912775c2cb52ece88c032c9c2464f544dd","observation_id":"963fe659-f831-48ff-a10a-51ae5926795a","resolution":{"observed_at":"2026-08-06T17:22:15.142692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-08-06T17:22:15.251884Z","title":"Opt: Open pre-trained transformer language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.251884Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:12e97b3ecaba995f6db8dee26f16661013f536d40ff51d4eeb134b15b5d52885","observation_id":"cb6a2be7-ccb1-4a1d-8b17-6225c4d86cb7","resolution":{"observed_at":"2026-08-06T17:22:15.251884Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:17.264977Z","title":"Clamp: Prompt-based contrastive learning for connecting language and animal pose","venue":null,"work_id":"eeff641c-f58a-44af-aa94-cc460801352b","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.316629Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:e824551e15b4c6e6c8b4285d0d3b7db5c59875e9112e5340e20648fa12ee46e6","observation_id":"54596103-3b07-4237-954f-a2cad0f014e8","resolution":{"observed_at":"2026-08-06T17:22:17.428420Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-06T17:22:15.365213Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.365213Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:54057c44f672137dc1380085523760d1af9318eb1933987626f487627341227d","observation_id":"f0e8c4af-a326-435b-ba64-3a26311a833e","resolution":{"observed_at":"2026-08-06T17:22:15.365213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T23:01:19.398301Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":66,"verified_exact":1,"verified_fuzzy":33},"total_outbound_references":108},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 100 of 108 outbound references and 0 inbound Pith citation observations for arXiv:2507.11102."}