{"as_of":"2026-08-07T05:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:807686cc2826577863ddd8ee4f08031c7396d4fc2edae0c1571a1f6818ea6dab","coverage":[{"denominator":108,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:22:15.365213Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.11102/citation-record","integrity":"/paper/2507.11102/integrity","json":"/paper/2507.11102/citation-record.json","paper":"/paper/2507.11102"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T17:22:08.397914Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.397914Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:8920e7ae43d706b5145ca763ad3093aca1e175a0656da8ecff87f3303a0f1b87","observation_id":"1e575dc4-462e-4c0d-8792-da60801519e8","resolution":{"observed_at":"2026-08-06T17:22:08.397914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.439737Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.439737Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:981999304ef196ff9f21c165d3f7cbaea6fc234d10b70043f3711d65472b9bf8","observation_id":"f21a9d29-8ac7-43af-8b26-d54d60991ba3","resolution":{"observed_at":"2026-08-06T17:22:08.439737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.500514Z","title":"2d human pose estimation: New benchmark and state of the art analysis","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.500514Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:31ff3c64a9c70f7c61cadc6a823d092c4900f536dadfc07634db30bedca71983","observation_id":"7207cbbb-8cd1-42d0-b7aa-1c872101389e","resolution":{"observed_at":"2026-08-06T17:22:08.500514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-06T17:22:08.549301Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.549301Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:1fe3491a8ddbb91c81cb78966fc92a061b42d5c7f37402dc84006a15dc0e9275","observation_id":"9af81807-c9b1-483e-a1f1-5bd556b02e07","resolution":{"observed_at":"2026-08-06T17:22:08.549301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.604576Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.604576Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:317c1d3b64ac857b846b192ab96b18728ecc24c33c1f50fa728325a2c6f7a25a","observation_id":"52b6fe1e-6910-40d6-9465-debb48ae05f9","resolution":{"observed_at":"2026-08-06T17:22:08.604576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.657396Z","title":"Cross-domain adaptation for animal pose estimation","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.657396Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:ab09b143257f01b67cbd8c19020288a0a9d6353bbf619d1069b8ad0b612f8cdc","observation_id":"3d5dcce1-3c34-4caf-bfc3-bc718347085b","resolution":{"observed_at":"2026-08-06T17:22:08.657396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15195","last_updated":"2023-07-03T16:08:00Z","snapshot_observed_at":"2026-07-06T15:47:07.545213Z","submitted_at":"2023-06-27T04:31:52Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15195","snapshot_observed_at":"2026-08-06T17:22:08.708494Z","title":"Shikra: Unleashing multimodal llm's referential dialogue magic","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.708494Z"},"links":{"cited_paper":"/paper/2306.15195","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:0ca49a446bded2e9b102db70755aa4ae3b91fb800dcdc628f505c8590a84d223","observation_id":"722e2f4f-6d8a-4e55-bd06-43be25a962dc","resolution":{"observed_at":"2026-08-06T17:22:08.708494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20340","last_updated":"2024-05-30T17:59:50Z","snapshot_observed_at":"2026-07-06T18:22:57.400982Z","submitted_at":"2024-05-30T17:59:50Z","title":"MotionLLM: Understanding Human Behaviors from Human Motions and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20340","snapshot_observed_at":"2026-08-06T17:22:08.784420Z","title":"Motionllm: Understanding human behaviors from human motions and videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.784420Z"},"links":{"cited_paper":"/paper/2405.20340","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:dd9b46b4b7dde71236a3305b012346ae7dc3a62ed43af94d3833627104a9483d","observation_id":"41e14b63-2c7c-442a-a5df-bdb37a9e620a","resolution":{"observed_at":"2026-08-06T17:22:08.784420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.873244Z","title":"Cascaded pyramid network for multi-person pose estimation","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.873244Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:158b0f5b4e84c1d5bb3a2ad818c0903fa5057b4cafce242fe5abc97cd1a0fae9","observation_id":"df60423c-1372-4f5a-a04d-352ff4cf9608","resolution":{"observed_at":"2026-08-06T17:22:08.873244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:08.937648Z","title":"Higherhrnet: Scale-aware representation learning for bottom-up human pose estimation","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:08.937648Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:6a14029a9fb073308ba63782ad6553c08e7bb1c3ff1a02823ccc71967c365473","observation_id":"ad8c833b-17f8-450f-87cb-5f33596beff0","resolution":{"observed_at":"2026-08-06T17:22:08.937648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.022376Z","title":"Palm: Scaling language modeling with pathways","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.022376Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:533d756b52b88bc045809029c07e36b23ad6aedb6523f4f34feffa0e5930d2ba","observation_id":"adab5577-926b-4db0-a236-928264f079de","resolution":{"observed_at":"2026-08-06T17:22:09.022376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.091426Z","title":"Instructblip: Towards general-purpose vision-language models with instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.091426Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:cec55e27dd1f93d17756d869a05a0fccbdf4e78c43e889574a0da5217c0f6f3e","observation_id":"81eae217-1435-45f6-aa0d-867eeb2773ad","resolution":{"observed_at":"2026-08-06T17:22:09.091426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.153335Z","title":"Model-agnostic meta-learning for fast adaptation of deep networks","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.153335Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:38b77f050ee1e88871bc55834a3272a83abd2f22bd551b1cb9ab96041cf8eaab","observation_id":"c30815ce-5e4e-4f70-85a7-6f2e35045019","resolution":{"observed_at":"2026-08-06T17:22:09.153335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.221988Z","title":"Deepfashion2: A versatile benchmark for detection, pose estimation, segmentation and re-identification of clothing images","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.221988Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:49263058dd3aff4f219533a1b4119a0bdf722048559cda007490ae4480eb5fb3","observation_id":"4273fe13-5bf9-4cbb-9c21-187d885d7cef","resolution":{"observed_at":"2026-08-06T17:22:09.221988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.03168","last_updated":"2025-02-28T14:39:17Z","snapshot_observed_at":"2026-07-06T18:40:57.362549Z","submitted_at":"2024-07-03T14:41:39Z","title":"LivePortrait: Efficient Portrait Animation with Stitching and Retargeting Control","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.03168","snapshot_observed_at":"2026-08-06T17:22:09.273146Z","title":"Liveportrait: Efficient portrait animation with stitching and retargeting control","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.273146Z"},"links":{"cited_paper":"/paper/2407.03168","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:a804965713bbcc02847adea9810e1a58db698d910514a111bb8ba87e2a7c63c8","observation_id":"63dbaaf5-b091-4f5f-912c-d04fd5006345","resolution":{"observed_at":"2026-08-06T17:22:09.273146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-06T17:22:09.347197Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.347197Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:32b8ba7cdbca906edf2a5ea56ccfed52802f89ddb6a91f51806988bcbe43f751","observation_id":"66112164-54e7-471d-afcc-d52f7409b51f","resolution":{"observed_at":"2026-08-06T17:22:09.347197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-06T17:22:09.421311Z","title":"Mistral 7b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.421311Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:516dbec17548955becbfca66515a1e1c058a830152b4fbe81670681e376304e2","observation_id":"e20c208d-7fd7-46dc-86ac-a750eb9ee0c6","resolution":{"observed_at":"2026-08-06T17:22:09.421311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.503131Z","title":"Multi-person articulated tracking with spatial and temporal embeddings","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.503131Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:c2ca0c1ac722485357a08974d6e2f72b38f4391ea962fc6a88668d177ce79dc9","observation_id":"edc05f21-d471-49ec-9075-97a12da249c4","resolution":{"observed_at":"2026-08-06T17:22:09.503131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.546326Z","title":"Differentiable hierarchical graph grouping for multi-person pose estimation","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.546326Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:f5d555fa0f98eb381e03928bd0112f0e6ef9165e5e56ec8bd8e8f3ea323b4f23","observation_id":"98051cca-7b82-4cd6-85fa-ce489709bbf6","resolution":{"observed_at":"2026-08-06T17:22:09.546326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.607010Z","title":"Whole-body human pose estimation in the wild","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.607010Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:bf49e6de6ed7497b0df3995851d6932cde6f323d08b3672eb43518b80f9cdb1a","observation_id":"28d9a720-b559-4054-85c2-507a171dcc03","resolution":{"observed_at":"2026-08-06T17:22:09.607010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.665516Z","title":"Human-art: A versatile human-centric dataset bridging natural and artificial scenes","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.665516Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:fd1b0fd9e57ff8ce3b033355aa6b705bd5c116147d0600cb0aede7d6a1b0395a","observation_id":"acb0fec6-1929-4787-8c39-1d07c9d5ca6f","resolution":{"observed_at":"2026-08-06T17:22:09.665516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.732831Z","title":"Humansd: A native skeleton-guided diffusion model for human image generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.732831Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:7445f54ec6954fddfad832749249fc80ed15417de246f18792e2e4339126f5c2","observation_id":"aa15c8a4-e594-438d-89e3-6b7be211d618","resolution":{"observed_at":"2026-08-06T17:22:09.732831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.795152Z","title":"Animalweb: A large-scale hierarchical dataset of annotated animal faces","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.795152Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:a417ebc70b7431ce9900e69622d9376b49f344cf73a0bdbf0f082611d6425f07","observation_id":"19cfd8aa-c4c9-4adf-aa5a-6c3e1a57a06f","resolution":{"observed_at":"2026-08-06T17:22:09.795152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.853259Z","title":"Segment anything","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.853259Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:9260070d397724258582af4c8fb301029ebd97bf8a1a665f0ab6cdfee4b45138","observation_id":"965e575e-c9ea-4af7-b542-be3f8a7b3ca7","resolution":{"observed_at":"2026-08-06T17:22:09.853259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:09.915013Z","title":"in the wild","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.915013Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:57649af6fd9be3a26057cc9222c6f74775216c5742fc14020efbd630594a0ba8","observation_id":"3e53591d-314e-4c55-a300-da9b080d313c","resolution":{"observed_at":"2026-08-06T17:22:09.915013Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.00692","last_updated":"2024-05-01T05:10:13Z","snapshot_observed_at":"2026-08-06T07:43:56.889679Z","submitted_at":"2023-08-01T17:50:17Z","title":"LISA: Reasoning Segmentation via Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.00692","snapshot_observed_at":"2026-08-06T17:22:09.963924Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:09.963924Z"},"links":{"cited_paper":"/paper/2308.00692","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:61da29557167dc188b8df0c531606557f0c42bfef8252fdc66ff116ce489ab46","observation_id":"096df928-9178-48e5-8a28-324a82b4db17","resolution":{"observed_at":"2026-08-06T17:22:09.963924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.02246","last_updated":"2024-05-03T17:00:00Z","snapshot_observed_at":"2026-07-06T18:09:29.876755Z","submitted_at":"2024-05-03T17:00:00Z","title":"What matters when building vision-language models?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.02246","snapshot_observed_at":"2026-08-06T17:22:10.040901Z","title":"What matters when building vision-language models? arXiv preprint arXiv:2405.02246, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.040901Z"},"links":{"cited_paper":"/paper/2405.02246","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:05dd61fdf288f1c04531d90fe85974f0827825cc1eae45d2f85695c296d4e8a6","observation_id":"41980a60-03c0-424b-914f-2f2352626e52","resolution":{"observed_at":"2026-08-06T17:22:10.040901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05425","last_updated":"2023-06-08T17:59:56Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"MIMIC-IT: Multi-Modal In-Context Instruction Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05425","snapshot_observed_at":"2026-08-06T17:22:10.082980Z","title":"Mimic-it: Multi-modal in-context instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.082980Z"},"links":{"cited_paper":"/paper/2306.05425","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:dfce20260858b1ba39775e16ff6b5945cac0192b06297018b6986ca2c57d57a5","observation_id":"e5a426ab-7b91-4d0e-8765-35979addef97","resolution":{"observed_at":"2026-08-06T17:22:10.082980Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.140423Z","title":"Crowdpose: Efficient crowded scenes pose estimation and a new benchmark","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.140423Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:f95188b12605857e43a136b821533695322cb985346865aa1b35d706154c4887","observation_id":"0831afbf-fc6c-4d6d-a7f9-89bacda92b36","resolution":{"observed_at":"2026-08-06T17:22:10.140423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.212644Z","title":"Human pose regression with residual log-likelihood estimation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.212644Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:a5cca11152770d1312d23803e70062f933a429cb2d82166ed58e607036a09d12","observation_id":"35498739-a94d-48f3-866a-72b9c207bf4d","resolution":{"observed_at":"2026-08-06T17:22:10.212644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.277017Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.277017Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:3d9a002977fa320c4b54e5cf7a7f6cfa4b1dbbcf7b3e216abce6a3e190039f00","observation_id":"7becd0fc-4cd5-4589-a604-332897af47e1","resolution":{"observed_at":"2026-08-06T17:22:10.277017Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.03516","last_updated":"2021-08-13T15:25:09Z","snapshot_observed_at":"2026-07-06T10:57:30.968546Z","submitted_at":"2021-04-08T05:12:38Z","title":"TokenPose: Learning Keypoint Tokens for Human Pose Estimation","version":3},"cited_work":{"arxiv_id":"2104.03516","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.03516","snapshot_observed_at":"2026-08-06T17:22:16.445473Z","title":"TokenPose: Learning Keypoint Tokens for Human Pose Estimation","venue":"cs.CV","work_id":"10845080-1946-4cbd-add1-e1dcb01ccbf6","year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.347059Z"},"links":{"cited_paper":"/paper/2104.03516","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:58eb4a24c603fd9f611345b3a3eb2c07347c55243aa7dd0e1e65832c9da73394","observation_id":"aca1c1be-834d-41ae-aeb5-8c43a9a3e5ca","resolution":{"observed_at":"2026-08-06T17:22:16.549635Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.392295Z","title":"Simcc: A simple coordinate classification perspective for human pose estimation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.392295Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:2b638d7632207a1a0ebcce9c3616433eefcad1e6d3d99239c73ae6b3c3ef7534","observation_id":"891c5c39-6db8-43bc-8402-b29843605d37","resolution":{"observed_at":"2026-08-06T17:22:10.392295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.18814","last_updated":"2024-03-27T17:59:04Z","snapshot_observed_at":"2026-07-31T05:41:28.385099Z","submitted_at":"2024-03-27T17:59:04Z","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.18814","snapshot_observed_at":"2026-08-06T17:22:10.455266Z","title":"Mini-gemini: Mining the potential of multi-modality vision language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.455266Z"},"links":{"cited_paper":"/paper/2403.18814","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:6d40ca87a9333ba9bff971f15b4df6640fd3a38c086e10d12e58a0398c11fed8","observation_id":"6069bdca-5dab-4ef4-a61d-a758b62f7f20","resolution":{"observed_at":"2026-08-06T17:22:10.455266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.07533","last_updated":"2024-05-16T21:21:30Z","snapshot_observed_at":"2026-07-06T17:00:37.507046Z","submitted_at":"2023-12-12T18:58:18Z","title":"VILA: On Pre-training for Visual Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.07533","snapshot_observed_at":"2026-08-06T17:22:10.522038Z","title":"Vila: On pre-training for visual language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.522038Z"},"links":{"cited_paper":"/paper/2312.07533","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:9c9f702639b337c225800f2ef6e2b14748e340b8cff2237504a4dbd307750e05","observation_id":"08768bc4-df80-43f8-a3c7-0cdc7de55359","resolution":{"observed_at":"2026-08-06T17:22:10.522038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.573970Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.573970Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:0f07e323525c29a68467d07a41d71013d46a3b572e9f721c306e7557747911d6","observation_id":"eefea3c9-1154-4999-9ecf-ae97a6a6659e","resolution":{"observed_at":"2026-08-06T17:22:10.573970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.659686Z","title":"Improved baselines with visual instruction tuning, 2023 a","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.659686Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:f83b848b0a93508dcb16ce0a10a1b9ffd2fbfbf9cf0900598a25ec991439a18d","observation_id":"29dfe3b6-5bb1-4e61-a502-9567b7b7ff88","resolution":{"observed_at":"2026-08-06T17:22:10.659686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.712260Z","title":"Visual instruction tuning, 2023 b","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.712260Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:fefa22d6c4146cdaf711384e6c62038a7665fae31de70e69274e3849f0c79c57","observation_id":"ae8ffabf-0df9-44f0-8646-2f4000136edc","resolution":{"observed_at":"2026-08-06T17:22:10.712260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.756326Z","title":"Llava-next: Improved reasoning, ocr, and world knowledge, January 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.756326Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:e87fbb45c839b2264ddac30bd2eb17c86825ba425cb1395c682e84280c739659","observation_id":"c9756c63-7999-4256-b83d-8bb282882a66","resolution":{"observed_at":"2026-08-06T17:22:10.756326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.818362Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.818362Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:4e2d7f3dcc523e8d41ce09cc3441dfc39dae9d00e3746fbc53b3c43ac9eb1253","observation_id":"f30433d3-3016-47b7-90e4-eea774095811","resolution":{"observed_at":"2026-08-06T17:22:10.818362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.875282Z","title":"A convnet for the 2020s","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.875282Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:a749e5ba978e3355a1679ae6b45f933510400258188c25c993bfbdc0b1609c07","observation_id":"daa8169b-1f48-4de4-ae48-fff91bc58e1a","resolution":{"observed_at":"2026-08-06T17:22:10.875282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:10.920288Z","title":"Deepseek-vl: Towards real-world vision-language understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.920288Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:0e72d4ac9f09ac6d1bbdd1d5a57db38ce1c12dd17667c6e1e937063b7f0449c0","observation_id":"939d3dcc-b1d7-4e5e-b825-3378cd0e8548","resolution":{"observed_at":"2026-08-06T17:22:10.920288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12978","last_updated":"2023-10-19T17:59:46Z","snapshot_observed_at":"2026-08-03T23:57:00.735076Z","submitted_at":"2023-10-19T17:59:46Z","title":"HumanTOMATO: Text-aligned Whole-body Motion Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12978","snapshot_observed_at":"2026-08-06T17:22:10.990900Z","title":"Humantomato: Text-aligned whole-body motion generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.990900Z"},"links":{"cited_paper":"/paper/2310.12978","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:d4994007d9349ab2f5505889401b909aeeb099fad8a38821ae16bce0b9842721","observation_id":"6e3305b5-61ea-4571-a060-fea05536bf93","resolution":{"observed_at":"2026-08-06T17:22:10.990900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.619568Z","title":"From keypoints to object landmarks via self-training correspondence: A novel approach to unsupervised landmark discovery","venue":null,"work_id":"e85f8db6-b6c8-4adc-a004-f1513937600a","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:10.997048Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:2bb1d86ee5ab7f4e4a4a72980d95c7b0af3a3c8a0ff96dcbeacf1fe8f4212e51","observation_id":"b6d07928-854b-4ce8-931b-60c564c9f891","resolution":{"observed_at":"2026-08-06T17:22:22.693855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.09611","last_updated":"2024-04-18T18:51:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-14T17:51:32Z","title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.09611","snapshot_observed_at":"2026-08-06T17:22:11.164532Z","title":"Mm1: Methods, analysis & insights from multimodal llm pre-training","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.164532Z"},"links":{"cited_paper":"/paper/2403.09611","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:a315b7440e0282ae28c89497a73fc9ce76d5c08b0ba34b9c23ecbec78d742e60","observation_id":"a5b53bb2-96bb-493c-b041-ddd1971f8bb5","resolution":{"observed_at":"2026-08-06T17:22:11.164532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08295","last_updated":"2024-04-16T12:52:47Z","snapshot_observed_at":"2026-08-03T03:29:01.959523Z","submitted_at":"2024-03-13T06:59:16Z","title":"Gemma: Open Models Based on Gemini Research and Technology","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08295","snapshot_observed_at":"2026-08-06T17:22:11.281720Z","title":"Gemma: Open models based on gemini research and technology","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.281720Z"},"links":{"cited_paper":"/paper/2403.08295","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:9f9df5c960e661a72deb5194dda88f1676af7f3eab85c17aaa75838ddb94fb6d","observation_id":"37e83a97-d2bc-4ebe-8f31-31181833f824","resolution":{"observed_at":"2026-08-06T17:22:11.281720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.450534Z","title":"Interhand2.6m: A dataset and baseline for 3d interacting hand pose estimation from a single rgb image","venue":null,"work_id":"4327c0ac-7f89-4614-ae1c-20fa76bb633a","year":2020},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.337039Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:ba37ffef241c0b133c5ec26c174745d466d3c302a5f3f5bf84585582ec1284a9","observation_id":"f3bb7e68-4f45-46ec-bb30-0a496f696331","resolution":{"observed_at":"2026-08-06T17:22:22.522833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.00216","last_updated":"2019-10-03T04:53:10Z","snapshot_observed_at":"2026-07-06T08:25:58.847575Z","submitted_at":"2019-10-01T06:21:50Z","title":"Revisiting Fine-tuning for Few-shot Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.00216","snapshot_observed_at":"2026-08-06T17:22:11.496851Z","title":"Revisiting fine-tuning for few-shot learning","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.496851Z"},"links":{"cited_paper":"/paper/1910.00216","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:d6e482ef1df6a6765e3d125a9e126cc429d2297f5c5e23adb376098a4c2f7899","observation_id":"9dba3041-d485-46d1-ab8e-c9ffaa0219d1","resolution":{"observed_at":"2026-08-06T17:22:11.496851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.355781Z","title":"Stacked hourglass networks for human pose estimation","venue":null,"work_id":"93cf2665-4d6c-499a-8b98-bd67ef78ad68","year":2016},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.582583Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:149c2f86b134f7b5e9e922038cdb65f0c37a2be39de6987a18d5b83c69918534","observation_id":"dc1ef1a2-29a6-48dd-92fa-dec6eceb1cdd","resolution":{"observed_at":"2026-08-06T17:22:22.402703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.248725Z","title":"Animal kingdom: A large and diverse dataset for animal behavior understanding","venue":null,"work_id":"63506b3c-6e3a-46bf-88f8-9cd0a39a9b1f","year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.692051Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:38b35c41591db7cbd1c67de2dc5f6368aee0454ef40d94ff877499165e4262ea","observation_id":"f6cdb89f-0137-4137-9b5e-dcc93f6750bb","resolution":{"observed_at":"2026-08-06T17:22:22.305728Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.167882Z","title":"Single-stage multi-person pose machines","venue":null,"work_id":"33ad63d2-333e-4afe-acd5-8b1424c0b82a","year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.803010Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:6e1fc2c2ea962ec291f9fae4d844546edacbac671f705a6fc255e340e0f6c371","observation_id":"9bacbe1e-dc12-4098-b34a-28ed5de5a620","resolution":{"observed_at":"2026-08-06T17:22:22.202969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-08-06T05:58:29.182448Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-06T17:22:11.891931Z","title":"Dinov2: Learning robust visual features without supervision","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:11.891931Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:f32830fbf58304fe8d9fc90e71666d9e5b0acd45f1325f717d989f72f5defcde","observation_id":"cec821b3-8e61-44ef-94e5-0bc0296fc372","resolution":{"observed_at":"2026-08-06T17:22:11.891931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03277","last_updated":"2023-04-06T17:58:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-06T17:58:09Z","title":"Instruction Tuning with GPT-4","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03277","snapshot_observed_at":"2026-08-06T17:22:12.010239Z","title":"Instruction tuning with gpt-4","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.010239Z"},"links":{"cited_paper":"/paper/2304.03277","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:fa35d394e379213f9c8d306d51d48cc8fc0afd35e9bb26de294fd0bf20d353fd","observation_id":"d5bacdeb-da28-496e-948a-7b8e26a9a10b","resolution":{"observed_at":"2026-08-06T17:22:12.010239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14824","last_updated":"2023-07-13T05:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-26T16:32:47Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14824","snapshot_observed_at":"2026-08-06T17:22:12.091020Z","title":"Kosmos-2: Grounding multimodal large language models to the world","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.091020Z"},"links":{"cited_paper":"/paper/2306.14824","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:c70ae067f53ac6d8a4f2f2c31e5bd552588695d25110445801874ffc54395129","observation_id":"2f35e897-2827-487e-a2d1-71341e8baec7","resolution":{"observed_at":"2026-08-06T17:22:12.091020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14167","last_updated":"2023-05-24T02:51:37Z","snapshot_observed_at":"2026-08-06T19:11:00.161773Z","submitted_at":"2023-05-23T15:37:28Z","title":"DetGPT: Detect What You Need via Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14167","snapshot_observed_at":"2026-08-06T17:22:12.156699Z","title":"Detgpt: Detect what you need via reasoning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.156699Z"},"links":{"cited_paper":"/paper/2305.14167","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:cdb34f396307cd8c9f44bd7e706bfa9b7ce6af79a45bfbee6e8482d42ade32a1","observation_id":"f0717450-6257-4e36-8e3a-1b0e092292d2","resolution":{"observed_at":"2026-08-06T17:22:12.156699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06612","last_updated":"2023-11-11T16:59:20Z","snapshot_observed_at":"2026-07-06T16:46:06.801011Z","submitted_at":"2023-11-11T16:59:20Z","title":"PerceptionGPT: Effectively Fusing Visual Perception into LLM","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.06612","snapshot_observed_at":"2026-08-06T17:22:12.222475Z","title":"Perceptiongpt: Effectively fusing visual perception into llm","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.222475Z"},"links":{"cited_paper":"/paper/2311.06612","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:1d9c6ff0f1c0e6a37ecf919b24e1abaa633aaf1d49aba956a0e2411eeff943bc","observation_id":"ac7378b4-d987-40b7-aad6-4158d99f5e3e","resolution":{"observed_at":"2026-08-06T17:22:12.222475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.093976Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"88d5a625-2a02-4551-8ead-088ac2e9d03f","year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.301281Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:070acba34a90e1f08f3edee0fcd63a9fce07e7c8c7c6056c535b5e161a7c50b5","observation_id":"00f6b4e1-e771-444b-8855-7ade893bbde3","resolution":{"observed_at":"2026-08-06T17:22:22.125639Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:22.001906Z","title":"Carfusion: Combining point tracking and part detection for dynamic 3d reconstruction of vehicles","venue":null,"work_id":"264d61a4-1416-448d-80f7-d64fc24fafb2","year":2018},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.360012Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:8c4d28e6d95bc17a47654495270870374a4a44b77c422ce4250a2cadb4ac815c","observation_id":"ab397d07-183e-4853-b753-63ea015c68a6","resolution":{"observed_at":"2026-08-06T17:22:22.053927Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.800841Z","title":"Zafeiriou, and M","venue":null,"work_id":"9477b7fe-c449-4713-b755-bc6f990f3dad","year":2016},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.443453Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:0df2257abc617799559b24e3693dc6b465c777a36ae386e23cb13007c9965c2d","observation_id":"c809e9d7-db83-40d6-82ee-37b313a06f21","resolution":{"observed_at":"2026-08-06T17:22:21.902725Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.632737Z","title":"Matching is not enough: A two-stage framework for category-agnostic pose estimation","venue":null,"work_id":"5eddc45c-a2db-41e0-aa0b-e4cc1104673c","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.518555Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:c75825ad16e5202b25b71a7cbfba2181f0961b34af0b0b5e90eb511d5234604a","observation_id":"193d7a0f-d084-44a4-b50e-334e49b1ad21","resolution":{"observed_at":"2026-08-06T17:22:21.697796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.516967Z","title":"Prototypical networks for few-shot learning","venue":null,"work_id":"cf851491-4f97-4eec-9f78-095d56483edf","year":2017},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.586243Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:725860e6fa6f8f87f4810dcaebe1c7cbe58eccc0ce6dbf4a6ae4fbb3a8358dae","observation_id":"0044afed-fb6e-4f5d-92b4-7bd38c1191e3","resolution":{"observed_at":"2026-08-06T17:22:21.562706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.408978Z","title":"Self-supervised keypoint discovery in behavioral videos","venue":null,"work_id":"0f50cdde-484b-4bdd-9996-1446d5386666","year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.645354Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:a0c383d2b3a8e1404f2d4e1c6bf027da1c1ed2475a7c78ba611f57408280c33f","observation_id":"d707b8cd-7738-4732-931a-b313d78d12ae","resolution":{"observed_at":"2026-08-06T17:22:21.473712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.289630Z","title":"Deep high-resolution representation learning for human pose estimation","venue":null,"work_id":"0b39eb2b-d11f-4e1a-a4ff-002e9de7f4fc","year":2019},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.715825Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:23995e16e16897ed902ef535fbe7cf7d4a3c0f290b3302623f162e9534ab5174","observation_id":"4a5fb059-5cab-44c9-8db4-85a02e3febdd","resolution":{"observed_at":"2026-08-06T17:22:21.322446Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:21.115262Z","title":"Compositional human pose regression","venue":null,"work_id":"2f0178d7-4b30-4d7c-8021-e8f60a610cb0","year":2017},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.772097Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:d28982b3f8d3d4fffb7b9cc8ddcc85b227ac398dab5c83c40e26997b51b6bbe8","observation_id":"9d19d238-f1ea-4518-a951-d71afbf21c20","resolution":{"observed_at":"2026-08-06T17:22:21.212839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:20.951726Z","title":"Deeppose: Human pose estimation via deep neural networks","venue":null,"work_id":"e1ecbc3f-0939-4a42-8c33-c33c12e01e65","year":2014},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.848054Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:05453d50db82c77acbe348255d03e79d05ad209f4074ebe2f32b9f462cca5a28","observation_id":"020a2b09-2843-435d-96e9-b60b71384085","resolution":{"observed_at":"2026-08-06T17:22:20.996690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-06T17:22:12.913739Z","title":"Llama: Open and efficient foundation language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:12.913739Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:ffe903f75a9dcc1e5573ab4e012795eb8e5d4de20e9c515fa178debfaed142a9","observation_id":"11fd6029-7466-4628-9867-b98b512596c8","resolution":{"observed_at":"2026-08-06T17:22:12.913739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-02T11:57:18.735747Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-06T17:22:13.017689Z","title":"Llama 2: Open foundation and fine-tuned chat models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.017689Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:974cada6e1ec7bd5d1c26ef7a2a76d317c456759117262c68b92ea34f3f950e0","observation_id":"ab97ce07-9584-4586-8107-d90835be4ba1","resolution":{"observed_at":"2026-08-06T17:22:13.017689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:20.835637Z","title":"Attention is all you need","venue":null,"work_id":"71e54be0-e7ce-47cf-8a1f-0381818ede5a","year":2017},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.097863Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:9aaf92c819ed036d38b26f5daad6b9eb896f7ffdc74657f125e2232eabdb4b65","observation_id":"c6e17968-63d8-4058-99ed-d85d64764396","resolution":{"observed_at":"2026-08-06T17:22:20.889936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:20.654754Z","title":"Locllm: Exploiting generalizable human keypoint localization via large language model","venue":null,"work_id":"60a84d01-b145-4248-84ad-32792c3599b8","year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.162173Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:ff776e92a2f735d8fe61652cc6f8ac5c84f6106a5d452662f5d7ac495505738b","observation_id":"1682b3c2-1b56-4d90-954f-4680153e4297","resolution":{"observed_at":"2026-08-06T17:22:20.760112Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:20.384593Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"852f2dd0-b846-453b-b670-dd71a634e705","year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.224523Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:31a5ec667bc69ae028eb573ef0e07a6d3035b9d9e11d1651b1c96815285cd086","observation_id":"1093694e-ff0d-42cc-b21b-0d6b8cc879ef","resolution":{"observed_at":"2026-08-06T17:22:20.495122Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:20.110723Z","title":"Convolutional pose machines","venue":null,"work_id":"07f6d653-32a9-4e81-973b-8fdef9f80301","year":2016},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.325920Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:34bfcd0a27cb11c341a7a861fe4976dbf77089fb1312552461a1bdb34270fae0","observation_id":"54aec589-a372-47f7-a2a5-8c701f4d0232","resolution":{"observed_at":"2026-08-06T17:22:20.249219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-08-06T17:22:13.366819Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model for hundreds of vision-language tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.366819Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:9ca980b5f7028580262834c38723bfb101a828ee17b9b3170d4a06848bcda32b","observation_id":"d43a7054-ee52-4ca0-8c0d-5cc729736004","resolution":{"observed_at":"2026-08-06T17:22:13.366819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05821","last_updated":"2025-04-11T14:21:24Z","snapshot_observed_at":"2026-07-06T18:27:50.189771Z","submitted_at":"2024-06-09T15:14:26Z","title":"F-LMM: Grounding Frozen Large Multimodal Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05821","snapshot_observed_at":"2026-08-06T17:22:13.424856Z","title":"F-lmm: Grounding frozen large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.424856Z"},"links":{"cited_paper":"/paper/2406.05821","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:5ca08506d37ddc0513c4f9a3f75e23a089439783f65c03ec77e9a5790297f30b","observation_id":"8029c7b6-d77d-49c4-a4dd-6d99ac6b047d","resolution":{"observed_at":"2026-08-06T17:22:13.424856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:19.924872Z","title":"Look at boundary: A boundary-aware face alignment algorithm","venue":null,"work_id":"88dff078-3c67-4abf-82af-7747e1015351","year":2018},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.517966Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:68753da33181b8c519f9876910b793ffb3f0850e797d543ea67ab1b6e48c990d","observation_id":"daa5e2e4-0e2d-460b-8b5b-bb93ef646c21","resolution":{"observed_at":"2026-08-06T17:22:19.991106Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:19.720960Z","title":"Simple baselines for human pose estimation and tracking","venue":null,"work_id":"1bed8c8e-b324-4283-bb00-cd25013fa5e1","year":2018},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.609050Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:1b1c160cddfcfa23cba0066a2d23f3355880d9936b0e775c46b16b0a9f9dff41","observation_id":"a8625e9d-9e14-4964-a116-79d40c55de5d","resolution":{"observed_at":"2026-08-06T17:22:19.816933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:19.503875Z","title":"Pixel-aligned language model","venue":null,"work_id":"70d343c1-43e0-4f3e-9be8-f115c844ad2e","year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.656756Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:31abf2cf68aad06553f3cba7a09c66bef2ab473fb2bbef3170d8ffba400d3ec6","observation_id":"4af29d8b-5f40-4ccb-91ba-6ff19c44019e","resolution":{"observed_at":"2026-08-06T17:22:19.613550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:19.183899Z","title":"Vipnas: Efficient video pose estimation via neural architecture search","venue":null,"work_id":"8f1bae23-34b1-4b25-a167-93c951e5be5c","year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.687889Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:33e6740777e464f1987c88fc8d232c7f4efafac64dbfeab5741182e9524e387b","observation_id":"1b189f69-544b-4dbb-b977-e3fc681299bf","resolution":{"observed_at":"2026-08-06T17:22:19.315519Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:19.014590Z","title":"Pose for everything: Towards category-agnostic pose estimation","venue":null,"work_id":"666799f4-2ba2-4f37-895a-f2f9e16b6c6d","year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.770149Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:0dcf8b9190712394850568d5d80ff66bfe13f773268e0b5fd8fced588ae0d1b0","observation_id":"f51d56c7-31ae-4ef5-986b-2b22187b2ddc","resolution":{"observed_at":"2026-08-06T17:22:19.077852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.858176Z","title":"Vitpose: Simple vision transformer baselines for human pose estimation","venue":null,"work_id":"612d2b96-3a85-43a7-9978-9ede8ce12966","year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.820520Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:5c6fc8c941d9546c2e85e1ac114fb3ade918bdc4e08b2b86e5b2b6907950e2d5","observation_id":"d665a2f6-f302-474a-9f64-27e72b9dfe03","resolution":{"observed_at":"2026-08-06T17:22:18.920419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12252","last_updated":"2023-05-20T17:59:23Z","snapshot_observed_at":"2026-07-06T15:30:05.820261Z","submitted_at":"2023-05-20T17:59:23Z","title":"Boosting Human-Object Interaction Detection with Text-to-Image Diffusion Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12252","snapshot_observed_at":"2026-08-06T17:22:13.857000Z","title":"Boosting human-object interaction detection with text-to-image diffusion model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.857000Z"},"links":{"cited_paper":"/paper/2305.12252","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:06287adaf1fd16ea1a9cf9dac1f5107670c5dd88cd51dbd7822a80c48a8e12c6","observation_id":"623d16ab-ef14-4aeb-8c7f-ac9bdd74423a","resolution":{"observed_at":"2026-08-06T17:22:13.857000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.729319Z","title":"Semantic human parsing via scalable semantic transfer over multiple label domains","venue":null,"work_id":"8ea9f98d-98cb-4bdc-8ff4-e34d34a53de7","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.935885Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:0d134372742f8808c5236704ec7e16e17cde94c8a997d471341676926eca3cc1","observation_id":"e48fbddf-4a94-4d53-982f-5612c5b30cd0","resolution":{"observed_at":"2026-08-06T17:22:18.797167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.607339Z","title":"Neural interactive keypoint detection","venue":null,"work_id":"57526ba3-090f-4571-9c5b-80fd970bd322","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:13.997849Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:24f0fc30acec9d159d3c39986456e4131b31a9980eccf29815042451f659f14b","observation_id":"d92d8528-f971-418f-befa-a03528242823","resolution":{"observed_at":"2026-08-06T17:22:18.660229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01593","last_updated":"2023-02-03T08:18:34Z","snapshot_observed_at":"2026-08-03T22:27:51.232439Z","submitted_at":"2023-02-03T08:18:34Z","title":"Explicit Box Detection Unifies End-to-End Multi-Person Pose Estimation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01593","snapshot_observed_at":"2026-08-06T17:22:14.093641Z","title":"Explicit box detection unifies end-to-end multi-person pose estimation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.093641Z"},"links":{"cited_paper":"/paper/2302.01593","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:22c383aed3b96ddae44c02869aa9aaff8252cbeecf62f12d055f8041522d86fa","observation_id":"d6f0255e-3426-4a8b-b49c-dedc33e2547f","resolution":{"observed_at":"2026-08-06T17:22:14.093641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08530","last_updated":"2024-07-17T09:25:24Z","snapshot_observed_at":"2026-07-06T16:32:07.186600Z","submitted_at":"2023-10-12T17:22:58Z","title":"X-Pose: Detecting Any Keypoints","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08530","snapshot_observed_at":"2026-08-06T17:22:14.151635Z","title":"Unipose: Detecting any keypoints","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.151635Z"},"links":{"cited_paper":"/paper/2310.08530","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:afec23a04fa273d603c9a062dca4f6e50b07521e808cb5d33b0403d9a9c5eae1","observation_id":"14404574-6dac-455e-868a-156d94e504ad","resolution":{"observed_at":"2026-08-06T17:22:14.151635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12435","last_updated":"2024-07-17T09:43:58Z","snapshot_observed_at":"2026-08-03T03:16:12.967109Z","submitted_at":"2024-07-17T09:43:58Z","title":"F-HOI: Toward Fine-grained Semantic-Aligned 3D Human-Object Interactions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.12435","snapshot_observed_at":"2026-08-06T17:22:14.216807Z","title":"F-hoi: Toward fine-grained semantic-aligned 3d human-object interactions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.216807Z"},"links":{"cited_paper":"/paper/2407.12435","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:8546a96c5f112c033151347c54f271d29b32d57e4a57b7dde768e8d5af967efd","observation_id":"fbdff743-f45d-4789-a08a-b3958541ca59","resolution":{"observed_at":"2026-08-06T17:22:14.216807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.452549Z","title":"Kptllm: Unveiling the power of large language model for keypoint comprehension","venue":null,"work_id":"c341664f-73e8-410b-bd01-e79cd0ae43a3","year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.279793Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:5350c4ca10ec28186db5923db3fdfcb833785f06e65d0638b43b836982eb5faa","observation_id":"4ba0d7ec-2233-4007-aaac-5d034fe7b636","resolution":{"observed_at":"2026-08-06T17:22:18.516527Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.294942Z","title":"Ed-pose++: Enhanced explicit box detection for conventional and interactive multi-object keypoint detection","venue":null,"work_id":"503f4101-3fc5-4a40-8879-b056042bab7f","year":2025},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.319317Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:34b0a14e7756ff7ce02942bb6837a120a9592c75e0597d8b2085a5e1f4b5c601","observation_id":"99992b3d-816a-4194-87e7-bafe0445bb99","resolution":{"observed_at":"2026-08-06T17:22:18.361369Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:18.151068Z","title":"Apt-36k: A large-scale benchmark for animal pose estimation and tracking","venue":null,"work_id":"ec6d6efe-573d-40e2-a3df-9460a4501948","year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.386205Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:86325ee426942e08cfd4ee755a3a6d7bc87ec7162b0144f75c9b8de7091eeb24","observation_id":"c17f125e-98d2-4054-8b8b-f13350b71b81","resolution":{"observed_at":"2026-08-06T17:22:18.199140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-06T17:22:14.478484Z","title":"mplug-owl: Modularization empowers large language models with multimodality","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.478484Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:54018c2f45b7473414c9931c846695e213d13f2df218609ba86f687b7c258c6d","observation_id":"ec2be9e1-6a25-49e9-92ff-b38306b2aec1","resolution":{"observed_at":"2026-08-06T17:22:14.478484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07704","last_updated":"2023-10-11T17:55:15Z","snapshot_observed_at":"2026-07-06T16:31:25.350087Z","submitted_at":"2023-10-11T17:55:15Z","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07704","snapshot_observed_at":"2026-08-06T17:22:14.559435Z","title":"Ferret: Refer and ground anything anywhere at any granularity","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.559435Z"},"links":{"cited_paper":"/paper/2310.07704","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:36d0fc2e333581532851c81019ff4de5e6382c93646e788074f2712e05977128","observation_id":"53a9ac90-37f5-488d-9b45-56ab876e9f8c","resolution":{"observed_at":"2026-08-06T17:22:14.559435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.12617","last_updated":"2021-11-01T05:36:12Z","snapshot_observed_at":"2026-07-06T11:42:16.189770Z","submitted_at":"2021-08-28T10:23:34Z","title":"AP-10K: A Benchmark for Animal Pose Estimation in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.12617","snapshot_observed_at":"2026-08-06T17:22:14.658414Z","title":"Ap-10k: A benchmark for animal pose estimation in the wild","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.658414Z"},"links":{"cited_paper":"/paper/2108.12617","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:b445f515c58182de3fa5472e8255f3afd895e26c4b99aec5e9a8d579cb8c4507","observation_id":"4bc9b194-deaa-45f1-95cb-ef08bbd1047d","resolution":{"observed_at":"2026-08-06T17:22:14.658414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.09408","last_updated":"2021-11-07T14:39:41Z","snapshot_observed_at":"2026-07-06T11:59:08.372025Z","submitted_at":"2021-10-18T15:37:58Z","title":"HRFormer: High-Resolution Transformer for Dense Prediction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.09408","snapshot_observed_at":"2026-08-06T17:22:14.730911Z","title":"Hrformer: High-resolution transformer for dense prediction","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.730911Z"},"links":{"cited_paper":"/paper/2110.09408","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:536ae3eed543ae2f6ff57a75af6b39c83abbcd8ca64a182465a42c6e065738c7","observation_id":"2d8289d9-32c5-4689-b2d0-f2de7b9836a2","resolution":{"observed_at":"2026-08-06T17:22:14.730911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18279","last_updated":"2024-08-12T07:14:00Z","snapshot_observed_at":"2026-08-06T08:08:30.426894Z","submitted_at":"2023-05-29T17:50:33Z","title":"Contextual Object Detection with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.18279","snapshot_observed_at":"2026-08-06T17:22:14.824478Z","title":"Contextual object detection with multimodal large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.824478Z"},"links":{"cited_paper":"/paper/2305.18279","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:cc08c9a2957bd00f4a369c8d7f7aa0d9ae036e67248dcc31ecb11788ba210562","observation_id":"23390d0a-6e3d-4aa9-8eb2-366838c9140e","resolution":{"observed_at":"2026-08-06T17:22:14.824478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:17.976686Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"d9057dff-7b3a-4d5c-9eee-b5f72b73ff38","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:14.952673Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:16f3ed18b15cdb5e7c10f9d8c900172a875da3bee7b1019a0a02d787041c69cf","observation_id":"1378ab1f-f0eb-418a-bf65-923ae2222a24","resolution":{"observed_at":"2026-08-06T17:22:18.057396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:17.825494Z","title":"Open-vocabulary animal keypoint detection with semantic-feature matching","venue":null,"work_id":"0d8bbf32-a9b1-4dc1-9558-27b712a103a1","year":2024},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.020088Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:05efbfb707e63afa3ec0d00ec71ab74dbefa273f39441ad922b578d37526538b","observation_id":"f921b1a8-6eff-4729-a310-922fc4c6a742","resolution":{"observed_at":"2026-08-06T17:22:17.934887Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:17.575802Z","title":"Adding conditional control to text-to-image diffusion models","venue":null,"work_id":"53206848-b1aa-4748-b6da-dff0865631c7","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.070625Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:13c89448630d1a50275f898cb06faf82fd9c44ad6965c26730133075beb35c01","observation_id":"4447a533-086c-48e3-a756-4810421bc257","resolution":{"observed_at":"2026-08-06T17:22:17.682866Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.03601","last_updated":"2025-06-12T00:15:18Z","snapshot_observed_at":"2026-08-06T08:24:07.376776Z","submitted_at":"2023-07-07T13:43:44Z","title":"GPT4RoI: Instruction Tuning Large Language Model on Region-of-Interest","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.03601","snapshot_observed_at":"2026-08-06T17:22:15.142692Z","title":"Gpt4roi: Instruction tuning large language model on region-of-interest","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.142692Z"},"links":{"cited_paper":"/paper/2307.03601","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:d28a79cdae29683718e45c19337adec9a06809581b2ce958315cb4791eef73d9","observation_id":"963fe659-f831-48ff-a10a-51ae5926795a","resolution":{"observed_at":"2026-08-06T17:22:15.142692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-08-06T17:22:15.251884Z","title":"Opt: Open pre-trained transformer language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.251884Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:f022fb8e18cf2b1bed2b35261987a53d80dd7c81477d5646af631a0233a58208","observation_id":"cb6a2be7-ccb1-4a1d-8b17-6225c4d86cb7","resolution":{"observed_at":"2026-08-06T17:22:15.251884Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:22:17.264977Z","title":"Clamp: Prompt-based contrastive learning for connecting language and animal pose","venue":null,"work_id":"eeff641c-f58a-44af-aa94-cc460801352b","year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.316629Z"},"links":{"citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:60e3a96056e3d6343110d8abc0dc34b8fa5ab984bf8d5cda014444e6808e7035","observation_id":"54596103-3b07-4237-954f-a2cad0f014e8","resolution":{"observed_at":"2026-08-06T17:22:17.428420Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-06T17:22:15.365213Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-06T17:22:15.365213Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2507.11102"},"observation_digest":"sha256:5d421d0c307648f16ab310bad8e26797c694bdd6d21486300dc9f005335f684b","observation_id":"f0e8c4af-a326-435b-ba64-3a26311a833e","resolution":{"observed_at":"2026-08-06T17:22:15.365213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.11102","last_updated":"2025-07-15T08:52:28Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T17:14:03.669426Z","submitted_at":"2025-07-15T08:52:28Z","title":"KptLLM++: Towards Generic Keypoint Comprehension with Large Language Model"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":66,"verified_exact":1,"verified_fuzzy":33},"total_outbound_references":108},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 100 of 108 outbound references and 0 inbound Pith citation observations for arXiv:2507.11102."}