{"as_of":"2026-08-15T01:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2bfccef9a0136515f63cef2b70b796865a5b0419c92ccb7259d96e596193d358","coverage":[{"denominator":50,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T12:58:09.284379Z","state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.01095/citation-record","integrity":"/paper/2509.01095/integrity","json":"/paper/2509.01095/citation-record.json","paper":"/paper/2509.01095"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.221785Z","title":"Posetrack: A benchmark for human pose estima- tion and tracking","venue":null,"work_id":"d45fd613-f62d-49d4-8c88-cd432c6d6bd3","year":null},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:04.031733Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:1d749a9b9b986b3c724d11d9a68750cab7e9332127e49ee0667dcd89c332ab3e","observation_id":"02dd0929-4571-4179-a82f-670d1b46186d","resolution":{"observed_at":"2026-08-05T12:58:10.227108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.204982Z","title":"Unipose: Unified hu- man pose estimation in single images and videos","venue":null,"work_id":"cd2ff066-1b0a-45c1-b95c-a2b6a4efba35","year":2020},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:04.090626Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:814218adec45f606c5b5dfb42e87cbdb21fef7efc4bf3a3e2aa8afd5950bf194","observation_id":"39082a6b-103d-4826-bb15-18c4ebdb69d9","resolution":{"observed_at":"2026-08-05T12:58:10.210118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.188557Z","title":"Pose-guided tracking-by-detection: Robust multi- person pose tracking","venue":null,"work_id":"9478d030-4f67-4772-9575-559678981902","year":2020},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:04.128578Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:41aa37b50993066c18cc1465c8867ff2b2c11232b4690b9c5159140d208805c4","observation_id":"dc8bc2d1-75b8-40c4-b3e8-8e492234b720","resolution":{"observed_at":"2026-08-05T12:58:10.193728Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.171512Z","title":"Tracking without bells and whistles","venue":null,"work_id":"343cb0fc-cdb4-41b6-876f-75439f6b0d8c","year":2019},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:04.180254Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:86e31725be2e35a9c32d836c35e78717711904ac909eed2c28af0714336ffca1","observation_id":"5f7781a0-0ac1-4f65-b101-b6a6cb926cd6","resolution":{"observed_at":"2026-08-05T12:58:10.176796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.146783Z","title":"Learning temporal pose esti- mation from sparsely-labeled videos","venue":null,"work_id":"ccb81732-24c9-46c4-83d5-4e0200cef9e7","year":2019},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:04.286179Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:4668140710d381d47555a9ab88a8760214ca4fcf9b6a292621a65e83386bdbbf","observation_id":"15f3ab81-e4e6-4a83-b4b7-f582d9cee22e","resolution":{"observed_at":"2026-08-05T12:58:10.151919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.129182Z","title":"End-to- end object detection with transformers","venue":null,"work_id":"a3f265bd-f578-4823-abf6-6522d94fddc2","year":2020},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:04.373293Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:3e4700865f1625f46151f2787404cde2021b5793f05b018c24c30dbddc1def00","observation_id":"d24ac1c8-6077-4dda-a8b9-88d251870aed","resolution":{"observed_at":"2026-08-05T12:58:10.134384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.111278Z","title":"Multi-context attention for hu- man pose estimation","venue":null,"work_id":"864abe1f-0f8a-4fee-871c-350974d79f6e","year":2017},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:04.461537Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:2ad8b9dd05c9ce044dc99dc3a9a677dac5be2770d5094f2f59090848f5432404","observation_id":"9fec910b-3b4e-413a-9085-90167c941de3","resolution":{"observed_at":"2026-08-05T12:58:10.116775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1805.04596","last_updated":"2018-07-20T13:04:02Z","snapshot_observed_at":"2026-08-14T19:16:13.918803Z","submitted_at":"2018-05-11T21:27:08Z","title":"Joint Flow: Temporal Flow Fields for Multi Person Tracking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1805.04596","snapshot_observed_at":"2026-08-05T12:58:04.553192Z","title":"Joint flow: Temporal flow fields for multi person tracking","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:04.553192Z"},"links":{"cited_paper":"/paper/1805.04596","citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:6c88f12dff016872a2144f350d0ba6b0c2901800404e69e9a11a08119a575256","observation_id":"8f09d8fe-c563-499b-bbd9-119a616d1add","resolution":{"observed_at":"2026-08-05T12:58:04.553192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.092299Z","title":"Posetrack21: A dataset for person search, multi-object tracking and multi-person pose tracking","venue":null,"work_id":"86708ad1-3ac9-473e-b3bc-58fc46c55401","year":2022},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:04.653772Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:83b4ebe1fa2ff03eb3336ab92130809b45b3e5339afa4cc8b6454c4773fbf3be","observation_id":"d8dc3638-c1de-4280-a82c-8dbb328fbd17","resolution":{"observed_at":"2026-08-05T12:58:10.098585Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.074658Z","title":"Rmpe: Regional multi-person pose estimation","venue":null,"work_id":"5b247353-5969-4884-9162-36820740c1ac","year":2017},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:04.753618Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:bc5bb3f2e062db7c8a360639ea7625a241029078ed4a04a036dd0e32cfe4e097","observation_id":"8b7b190d-f5ae-4fb7-8b7f-4e16f33fb7d6","resolution":{"observed_at":"2026-08-05T12:58:10.080124Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.056965Z","title":"Detect-and-track: Efficient pose estimation in videos","venue":null,"work_id":"0829b069-82a3-409c-81e9-0814a5d18168","year":2018},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:04.898016Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:290dcbf031ae62db8adc9f2640c318a9e5f0e6e0af311e8b8f43d8f12d0e2bb7","observation_id":"b2fa4981-77be-4d04-8673-11563c23f92f","resolution":{"observed_at":"2026-08-05T12:58:10.062167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.040279Z","title":"Multi-domain pose net- work for multi-person pose estimation and tracking","venue":null,"work_id":"cba99934-5a0e-4290-9580-db4e958073e8","year":2018},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:05.064082Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:6e8277d34fc14909c7e89186a10e72ef7b2b1dfccef429208ef08ea8c6b12428","observation_id":"2e6ee975-9a0f-4cce-84a1-f0625f1d096e","resolution":{"observed_at":"2026-08-05T12:58:10.045551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.022355Z","title":"Pose estimator and tracker using temporal flow maps for limbs","venue":null,"work_id":"b2567edd-6443-4f97-a4d0-4415b7d6db67","year":2019},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:05.192184Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:af137c08ea927e55c4e38eb18963f6fdc11bef778b0831747b7d458c75bc7c7a","observation_id":"12e24822-7d34-4097-98d1-65bb5c23a343","resolution":{"observed_at":"2026-08-05T12:58:10.027147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:10.003588Z","title":"Posetrack: Joint multi-person pose estimation and tracking","venue":null,"work_id":"9d3b9464-f60a-4da2-bc25-f42058e861d4","year":2011},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:05.346399Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:394c997a3504492de843cca51595a444e6f0cb324e251b4a6a3dc182e02ef1a7","observation_id":"c650bc4d-dc04-4860-ad31-2aa09d7ac510","resolution":{"observed_at":"2026-08-05T12:58:10.009549Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.07399","last_updated":"2023-07-03T03:06:26Z","snapshot_observed_at":"2026-08-13T12:25:27.314050Z","submitted_at":"2023-03-13T18:26:11Z","title":"RTMPose: Real-Time Multi-Person Pose Estimation based on MMPose","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.07399","snapshot_observed_at":"2026-08-05T12:58:05.487775Z","title":"Rtmpose: Real- time multi-person pose estimation based on mmpose","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:05.487775Z"},"links":{"cited_paper":"/paper/2303.07399","citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:ed0f1ecf1fc0d0ea0f0c45690fb6044065f153d3f0f2e110eafa2425a75212b1","observation_id":"3eea993f-54b9-4db4-9284-d81c7e32fe56","resolution":{"observed_at":"2026-08-05T12:58:05.487775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.985034Z","title":"Multi-person articulated tracking with spatial and tempo- ral embeddings","venue":null,"work_id":"701c8a37-2238-4c31-881b-9a8a0664fe9a","year":2019},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:05.648577Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:ad520d28f85a6299baec4f88f9202b295e148c206c8416b4b67b557cedca4c57","observation_id":"3c5e4d58-d90a-45ba-b20e-e87685a1cd5a","resolution":{"observed_at":"2026-08-05T12:58:09.990821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.967204Z","title":"The hungarian method for the assignment problem","venue":null,"work_id":"b2f7eb29-a28e-42a4-8128-9e5d9f33a5ea","year":null},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:05.793918Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:717338e8d49560c84db8372afa9650ad3d493a3d97712a872a63954f497a36ad","observation_id":"d4dae296-5e04-427c-a5e1-3712fa240daa","resolution":{"observed_at":"2026-08-05T12:58:09.972361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.950168Z","title":"Simcc: A simple coordinate classification perspective for hu- man pose estimation","venue":null,"work_id":"543d4b2f-360a-4a93-9985-819724506a54","year":2022},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:06.005030Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:3a5a45f37816a44577baa0445f12932fa0d70ed6186adebf68f370bba2b039ac","observation_id":"489503b5-0edb-4b91-87c6-5cfe20d93d27","resolution":{"observed_at":"2026-08-05T12:58:09.955645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.930957Z","title":"Group pose: A simple baseline for end-to- end multi-person pose estimation","venue":null,"work_id":"54ee1ca0-04c1-43e1-b9fe-2fc56c3ef935","year":2023},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:06.127376Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:ef71626a6466ecd161be4fc975c4cc3056a6a3456aa5549b3a7592e112e28086","observation_id":"30d85180-f478-4ca1-8834-36067123a302","resolution":{"observed_at":"2026-08-05T12:58:09.936231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.913688Z","title":"Towards natural and accurate future motion prediction of humans and animals","venue":null,"work_id":"6611d6af-edc0-4c0a-aa40-928c59df4d87","year":2019},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:06.194909Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:10e7ef86114e89c3eaf0014ec8583f93bf4c59005d09968452a1ea5d2482b443","observation_id":"21ba59f8-7ad6-4b58-8471-710a0be68415","resolution":{"observed_at":"2026-08-05T12:58:09.918873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.897295Z","title":"Deep dual consec- utive network for human pose estimation","venue":null,"work_id":"fb51ed35-8705-45be-ab14-9c2083d482a3","year":2021},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:06.259248Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:d52d394d98e09b9bd3a3d3428b1056da8ab315fdd3b0a4b2930dde5ebd90d981","observation_id":"8453b2fe-bd31-4912-aafb-2462c76fb9a3","resolution":{"observed_at":"2026-08-05T12:58:09.902179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.880284Z","title":"Tempo- ral feature alignment and mutual information maximization for video-based human pose estimation","venue":null,"work_id":"dd1a53d8-6688-49f7-917f-01b23c0ae131","year":2022},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:06.360747Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:8ca2596347deba6c5620fb1f5727d68ef6d0be5003fc5f500d319dd0f7d00c21","observation_id":"562af579-3f80-4f07-b23d-33e438487f39","resolution":{"observed_at":"2026-08-05T12:58:09.885966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.863555Z","title":"Lstm pose ma- chines","venue":null,"work_id":"30b01822-a25a-4cfd-8039-c3dce00950c7","year":2018},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:06.468666Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:81f24299f1f56d828af02922adc88e3bc7252d46031e92ffdc4543d7d811aa7f","observation_id":"233153fc-3e2d-42e4-946a-f5c4d3c7587a","resolution":{"observed_at":"2026-08-05T12:58:09.868336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.846014Z","title":"Stacked hour- glass networks for human pose estimation","venue":null,"work_id":"55bd2b66-edc4-432c-bec3-09118b2c2e5c","year":2016},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:06.618167Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:a19e20a9b58c3b01c46c22c716ed805f203dff6d84ee9edb174f5fe7d5bbf34e","observation_id":"f5aa63a5-7d0d-425a-8d3f-8c3c37c1a947","resolution":{"observed_at":"2026-08-05T12:58:09.852068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.829559Z","title":"Flow- ing convnets for human pose estimation in videos","venue":null,"work_id":"36f238b2-ed6d-40f8-8976-50107d4f3304","year":1913},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:06.710124Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:ca078fca4d6f5123580716462b86dbbf8e40e08dbbaf02d976d4cd4119cd99a2","observation_id":"a6aeb5a7-d70b-45d2-a3f8-cc32a37dd528","resolution":{"observed_at":"2026-08-05T12:58:09.834552Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.813148Z","title":"Efficient online multi-person 2d pose tracking with recurrent spatio-temporal affinity fields","venue":null,"work_id":"44f098ee-ef0e-40e2-b2a4-6a014f22dafd","year":2019},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:06.805473Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:bf1508ec8ccfa0a4b561f677259c4398a7477eeaf963fbf0b8822b3dac888962","observation_id":"dd251701-da83-4ed8-8e82-d08789df523f","resolution":{"observed_at":"2026-08-05T12:58:09.818205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.795960Z","title":"Self-supervised keypoint correspondences for multi- person pose estimation and tracking in videos","venue":null,"work_id":"2ef76e56-bd83-4745-b5cd-171d046f6b70","year":null},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:06.919024Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:cb2226ec8dcfb2076a9ba17f84dac33e8e5b78a6dd89c21440d34a221545ec40","observation_id":"270bbbc5-23de-46be-9327-d0ff946daa6e","resolution":{"observed_at":"2026-08-05T12:58:09.801092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1804.02767","last_updated":"2018-04-08T22:27:57Z","snapshot_observed_at":"2026-08-06T11:09:16.409556Z","submitted_at":"2018-04-08T22:27:57Z","title":"YOLOv3: An Incremental Improvement","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1804.02767","snapshot_observed_at":"2026-08-05T12:58:07.031861Z","title":"Yolov3: An incremental improvement","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:07.031861Z"},"links":{"cited_paper":"/paper/1804.02767","citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:009a0b600a65d8585f6ef4e465640bd6bdc5c7a88901671054217272162ece7a","observation_id":"caf5c5e1-c06f-460d-9c7e-59ff3a27fa60","resolution":{"observed_at":"2026-08-05T12:58:07.031861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.777365Z","title":"End-to-end multi-person pose estimation with transformers","venue":null,"work_id":"cd57b611-37ac-4f50-9511-1eeba7c5f628","year":2022},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:07.159566Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:da0b985919999cad16c7f5b33507b0cfa19033ec34cabd3988fb67e793768bdf","observation_id":"6efa9863-106a-4c65-a5de-d0ffc53b0335","resolution":{"observed_at":"2026-08-05T12:58:09.783014Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.758398Z","title":"Thin-slicing network: A deep structured model for pose esti- mation in videos","venue":null,"work_id":"13531022-6b96-43e3-bfab-b424567b0b94","year":null},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:07.258155Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:72ae481440477569fee5ca30146fff8fc89193efd89c7d76b7b0e665fcc0f027","observation_id":"008a22ea-190c-4613-b866-8b7c049682e3","resolution":{"observed_at":"2026-08-05T12:58:09.764389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.739901Z","title":"Deep high-resolution representation learning for human pose esti- mation","venue":null,"work_id":"55fe93a6-64c2-457d-93a3-917033eee281","year":2019},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:07.374311Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:6a7cb2290859140fd1727584c6d9f43823139ba80c1dc2f54e6fc30f5b69dcc4","observation_id":"49c16f32-3830-4925-9378-cc98ed617c5d","resolution":{"observed_at":"2026-08-05T12:58:09.745119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.723769Z","title":"Sparse r-cnn: End-to-end ob- ject detection with learnable proposals","venue":null,"work_id":"6d899f72-50aa-4077-abb9-b15626caa6e9","year":2021},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:07.563755Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:923937b3eb8fce8a2ab3efa8238bf2474041c91dd844f72ac5026acb047b3669","observation_id":"d4584491-19bb-497e-9bae-5a49f3435d93","resolution":{"observed_at":"2026-08-05T12:58:09.728577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.706422Z","title":"Deeppose: Human pose estimation via deep neural networks","venue":null,"work_id":"1f84c1fa-181b-4796-a1d8-2ad258cff0b0","year":2014},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:07.685371Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:51789101417ae6a97582be813d118f1db1607094ca8732ea261c81d95fbae978","observation_id":"d929ec27-2ec6-4ef0-9362-06d3c373e323","resolution":{"observed_at":"2026-08-05T12:58:09.712125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.688980Z","title":"Attention is all you need","venue":null,"work_id":"53761a81-8ead-4d2a-9e5c-55de45385860","year":2017},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:07.854190Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:74a54edf79dd675f2746e820282e452a626583fcbc879fe87849c62de8d31bb4","observation_id":"85a90c97-6630-49ae-b4c0-ba030624af14","resolution":{"observed_at":"2026-08-05T12:58:09.694785Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.670056Z","title":"Beyond physical connections: Tree models in human pose estimation","venue":null,"work_id":"3f67751f-3c3a-41b4-a63b-209aeb0e2bb9","year":2013},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:07.976687Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:7a02383d14fe6cfb3ddb2c24a0527940638d87aa539a476db9c0e1a30d536b08","observation_id":"1b0111e7-b0b6-471a-8f4d-450e85cf7876","resolution":{"observed_at":"2026-08-05T12:58:09.676477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.652103Z","title":"Deep high-resolution represen- tation learning for visual recognition","venue":null,"work_id":"6c319db0-6d3a-4e71-b524-bd5df1b8b1a8","year":2020},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:08.136487Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:de5cfbdadb135c665dc443bb7dc6911ffa5696ed4b7610dd531d4e3d025334d7","observation_id":"bc2f530e-e2d6-4497-8eee-75480098bfe1","resolution":{"observed_at":"2026-08-05T12:58:09.657534Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.634820Z","title":"Com- bining detection and tracking for human pose estimation in videos","venue":null,"work_id":"a4377c24-3acb-48eb-b4b4-198664c5df62","year":2020},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:08.296891Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:117ac4dcce1d48a0bb3e1ef777eb459ed0dc9b216093dd2176a56dc754996bb2","observation_id":"3a5c5c33-8c2a-4fc3-82ee-b69c585aa01d","resolution":{"observed_at":"2026-08-05T12:58:09.640034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.618245Z","title":"Convolutional pose machines","venue":null,"work_id":"cdd04be5-f96a-48e8-a720-21f8d9ebdb9c","year":2016},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:08.403860Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:8fb6739256eed0af84fbfca01de14fcd5109cb9d98a38d37dd6a717759156e9f","observation_id":"f33487db-c180-4800-b90b-69d407e21358","resolution":{"observed_at":"2026-08-05T12:58:09.623208Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.599419Z","title":"Simple baselines for human pose estimation and tracking","venue":null,"work_id":"746b69c9-c291-410f-9b61-932f440784e8","year":2018},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:08.471164Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:85e8c113942ed777fc420e19422636c6fbcc0a4667bdcfd80c86a3bf21355d2f","observation_id":"71a8ec58-7b59-46df-af75-8399f109a205","resolution":{"observed_at":"2026-08-05T12:58:09.604740Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.581373Z","title":"Simple baselines for human pose estimation and tracking","venue":null,"work_id":"b5c654c6-e467-4923-b6c3-4acca4930393","year":2018},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:08.578988Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:c5408ce9e635edbebfd94798edc541dc30466846a8696c43e920eea05bd30834","observation_id":"90208266-81d6-499b-b328-8386e41ae3ac","resolution":{"observed_at":"2026-08-05T12:58:09.587136Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.562125Z","title":"Querypose: Sparse multi- person pose regression via spatial-aware part-level query","venue":null,"work_id":"bb846476-9fac-495c-a16a-34ad5a058c23","year":2022},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:08.736263Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:61c81b3532cf743f6ba35dd7cfd265345c1dfd44fefac5e1bf91efabfd7cd027","observation_id":"410093a1-ba70-4ede-b837-a19ddf104061","resolution":{"observed_at":"2026-08-05T12:58:09.568639Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.00977","last_updated":"2018-07-02T18:40:46Z","snapshot_observed_at":"2026-08-14T19:49:17.824751Z","submitted_at":"2018-02-03T14:08:36Z","title":"Pose Flow: Efficient Online Pose Tracking","version":2},"cited_work":{"arxiv_id":"1802.00977","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.00977","snapshot_observed_at":"2026-08-05T12:58:09.392336Z","title":"Pose Flow: Efficient Online Pose Tracking","venue":"cs.CV","work_id":"c88a830f-8b1b-4360-9db3-0b86ff66711f","year":2018},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:08.889777Z"},"links":{"cited_paper":"/paper/1802.00977","citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:9a5b905f85ce5ce40d9c7d6d918ba3fefe30c1d1d184a8df0f15165355cabb79","observation_id":"83bb42e7-c4a9-4876-b37e-22163ad5418b","resolution":{"observed_at":"2026-08-05T12:58:09.398202Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.541413Z","title":"Vit- pose: Simple vision transformer baselines for human pose estimation","venue":null,"work_id":"b9e0ccf6-18eb-4be0-baec-9bd70e965a76","year":2022},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:09.008425Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:63ddaa5deefdb6c4c1529bebacb6f62fc2078f32cb18d556a5d76c71629ab125","observation_id":"de82c7dd-e9c3-4fa2-9a2d-b93b493a6105","resolution":{"observed_at":"2026-08-05T12:58:09.547129Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.523054Z","title":"Spatial tempo- ral graph convolutional networks for skeleton-based action recognition","venue":null,"work_id":"a431f30e-2e91-48b7-a4fd-9cb1aa918fca","year":2018},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:09.071770Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:daad8494c2d41042d5c00205770e0bae7b6e1c629a9fe9d7f1c652d8f5ff1294","observation_id":"1d6ddcb8-19fa-460c-9e62-e081f34675b4","resolution":{"observed_at":"2026-08-05T12:58:09.528344Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01593","last_updated":"2023-02-03T08:18:34Z","snapshot_observed_at":"2026-08-13T12:52:13.526480Z","submitted_at":"2023-02-03T08:18:34Z","title":"Explicit Box Detection Unifies End-to-End Multi-Person Pose Estimation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01593","snapshot_observed_at":"2026-08-05T12:58:09.185540Z","title":"Explicit box detection unifies end-to-end multi-person pose estimation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:09.185540Z"},"links":{"cited_paper":"/paper/2302.01593","citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:ca659ee99dcd123ff6c3771fe9ba69bf52d3653b8d146686b7808415d87dbd64","observation_id":"cd687bf4-3cb2-44ad-94d1-12cd0fa5657f","resolution":{"observed_at":"2026-08-05T12:58:09.185540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.504181Z","title":"Trans- pose: Keypoint localization via transformer","venue":null,"work_id":"84de15ea-6121-4b48-82ab-ea844488e5ff","year":2021},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:09.235896Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:cf1cec9cddbb26fa0521abdf21a858b1d7d50712da4e11125c174762d90e62a3","observation_id":"829bb7d3-5fa2-4b3f-b95f-20b3d83ea1bc","resolution":{"observed_at":"2026-08-05T12:58:09.509908Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.486573Z","title":"Learning dynamics via graph neural networks for human pose estimation and tracking","venue":null,"work_id":"14135bde-0b92-4375-a5dc-2c74a15a67d0","year":2021},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:09.269113Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:2b9c0caf5872f14303c7024392e3061a7b831ccfbfd9d9a8c42be8929bf8fe1a","observation_id":"ec425290-6dc8-42bb-a480-3601fc856e57","resolution":{"observed_at":"2026-08-05T12:58:09.492003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.05593","last_updated":"2019-08-15T15:42:57Z","snapshot_observed_at":"2026-08-14T22:42:43.340378Z","submitted_at":"2019-08-15T15:42:57Z","title":"FastPose: Towards Real-time Pose Estimation and Tracking via Scale-normalized Multi-task Networks","version":1},"cited_work":{"arxiv_id":"1908.05593","doi":null,"metadata_source":"pith","pith_arxiv_id":"1908.05593","snapshot_observed_at":"2026-08-05T12:58:09.344972Z","title":"FastPose: Towards Real-time Pose Estimation and Tracking via Scale-normalized Multi-task Networks","venue":"cs.CV","work_id":"d6bbf9fa-3c79-491b-974e-9e98eabef2b0","year":2019},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:09.274138Z"},"links":{"cited_paper":"/paper/1908.05593","citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:d12a6295be20a39542a5cf14d7c1ee4ebf7d47c037232dde506c6bd938a74871","observation_id":"5a1ba258-6c25-4af1-89f5-6e8a98cf6508","resolution":{"observed_at":"2026-08-05T12:58:09.352310Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:58:09.468788Z","title":"Efficient human pose estimation via parsing a tree structure based human model","venue":null,"work_id":"56e32861-485d-459d-b375-5269f6ef4dcc","year":2009},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:09.279546Z"},"links":{"citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:efc7a952e197fe93bf2162780a844f5dcdf5c17263ece0066fa102c44d53ae89","observation_id":"61ddf8ee-a379-4570-9b7a-110697db9c3c","resolution":{"observed_at":"2026-08-05T12:58:09.474016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.04159","last_updated":"2021-03-18T03:14:26Z","snapshot_observed_at":"2026-08-13T17:54:01.724774Z","submitted_at":"2020-10-08T17:59:21Z","title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.04159","snapshot_observed_at":"2026-08-05T12:58:09.284379Z","title":"Deformable detr: Deformable trans- formers for end-to-end object detection","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-05T12:58:09.284379Z"},"links":{"cited_paper":"/paper/2010.04159","citing_paper":"/paper/2509.01095"},"observation_digest":"sha256:85d2cf646d39dbbd0e8073f36040a691724fd6c5276e7fafe157dbd15fc383ad","observation_id":"5f59f41f-9587-4e68-abd6-712f6eac0d79","resolution":{"observed_at":"2026-08-05T12:58:09.284379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.01095","last_updated":"2025-09-01T03:34:57Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T22:43:45.895416Z","submitted_at":"2025-09-01T03:34:57Z","title":"An End-to-End Framework for Video Multi-Person Pose Estimation"},"reference_resolution":{"displayed":50,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":5,"verified_exact":2,"verified_fuzzy":43},"total_outbound_references":50},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 50 of 50 outbound references and 0 inbound Pith citation observations for arXiv:2509.01095."}