{"as_of":"2026-08-20T02:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b22f42cbfc243b71022513d6b5fee957080e759ddc24ef93aef409b3f37be3b1","coverage":[{"denominator":17,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T05:46:49.074505Z","state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2504.19847/citation-record","integrity":"/paper/2504.19847/integrity","json":"/paper/2504.19847/citation-record.json","paper":"/paper/2504.19847"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:46:49.286645Z","title":null,"venue":null,"work_id":"684e7b32-2190-4d24-abb3-35ba91418d15","year":2023},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.006517Z"},"links":{"citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:0e93cd7b378857ecc8822058ba45505c190cafdf0b8d26f373949bd6504930d3","observation_id":"02ea5475-83e0-437c-8ff2-ce56285ddb16","resolution":{"observed_at":"2026-08-16T05:46:49.290650Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.00571","last_updated":"2023-11-01T15:13:43Z","snapshot_observed_at":"2026-08-16T14:46:51.110409Z","submitted_at":"2023-11-01T15:13:43Z","title":"LLaVA-Interactive: An All-in-One Demo for Image Chat, Segmentation, Generation and Editing","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.00571","snapshot_observed_at":"2026-08-16T05:46:49.011495Z","title":"arXiv preprint arXiv:2311.00571","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.011495Z"},"links":{"cited_paper":"/paper/2311.00571","citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:4136e19916d6f332658a913dda44e86d43278e0ee56c96bcf95f31cda1e51a32","observation_id":"7f851866-9bdf-43f3-a7f7-00b6dc322be3","resolution":{"observed_at":"2026-08-16T05:46:49.011495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1808.10437","last_updated":"2018-08-30T17:58:33Z","snapshot_observed_at":"2026-08-19T21:15:28.099883Z","submitted_at":"2018-08-30T17:58:33Z","title":"iCAN: Instance-Centric Attention Network for Human-Object Interaction Detection","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1808.10437","snapshot_observed_at":"2026-08-16T05:46:49.021627Z","title":"arXiv preprint arXiv:1808.10437","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.021627Z"},"links":{"cited_paper":"/paper/1808.10437","citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:d441ff525db76b81c8dba5089d948bfdf55e809bafa7075dde77f1fa88ca4720","observation_id":"8ff6e376-9cc5-4927-acdf-669bf89f9d46","resolution":{"observed_at":"2026-08-16T05:46:49.021627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:46:49.274578Z","title":null,"venue":null,"work_id":"46cd84a8-aaf2-40df-b83a-5a758b878905","year":2020},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.031370Z"},"links":{"citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:a82e68bfcc6a77beae380c472cd0ca6c658990b38b195d29025353d63b948a9f","observation_id":"b8f8ea44-11ed-4612-872d-cb1f2a7d6cc1","resolution":{"observed_at":"2026-08-16T05:46:49.278559Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-08-20T02:22:47.212760Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-16T05:46:49.045303Z","title":"arXiv preprint arXiv:2305.05662","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.045303Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:2aecf5f980e59410c36205a3ecba609b0d6eed83af5fb22d21bcd4518dcae841","observation_id":"53607f26-2daf-445b-b3fa-b9201fa7bfed","resolution":{"observed_at":"2026-08-16T05:46:49.045303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16524","last_updated":"2024-04-08T15:46:09Z","snapshot_observed_at":"2026-08-16T14:56:48.566171Z","submitted_at":"2023-09-28T15:34:49Z","title":"HOI4ABOT: Human-Object Interaction Anticipation for Human Intention Reading Collaborative roBOTs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16524","snapshot_observed_at":"2026-08-16T05:46:49.055521Z","title":"arXiv preprint arXiv:2309.16524","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.055521Z"},"links":{"cited_paper":"/paper/2309.16524","citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:528c3d388c5e9af39d41e8f7cd6a0a96c7708d7625b1c79b3a87313d7e8c26d2","observation_id":"5b19f973-773e-4c95-af69-7112db5b5483","resolution":{"observed_at":"2026-08-16T05:46:49.055521Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.03605","last_updated":"2022-07-11T10:30:29Z","snapshot_observed_at":"2026-08-18T07:50:02.322273Z","submitted_at":"2022-03-07T18:55:26Z","title":"DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.03605","snapshot_observed_at":"2026-08-16T05:46:49.069455Z","title":"19717–19728","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.069455Z"},"links":{"cited_paper":"/paper/2203.03605","citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:8c252a410dffea7785f3fe614ea1c4963773f5ac3b6fb30c8e7fa472e322be97","observation_id":"99ae6d95-c307-4ba6-bb05-3a60f445332c","resolution":{"observed_at":"2026-08-16T05:46:49.069455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.04159","last_updated":"2021-03-18T03:14:26Z","snapshot_observed_at":"2026-08-13T17:54:01.724774Z","submitted_at":"2020-10-08T17:59:21Z","title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.04159","snapshot_observed_at":"2026-08-16T05:46:49.074505Z","title":"arXiv preprint arXiv:2010.04159","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.074505Z"},"links":{"cited_paper":"/paper/2010.04159","citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:ae3a940c08fe69d281e67bc870758d6fc2607394d30c2bd6bf2ca0742b3f53ac","observation_id":"b97a4b0f-ea87-46a8-81cd-e7c9be54cc2f","resolution":{"observed_at":"2026-08-16T05:46:49.074505Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:46:49.262914Z","title":null,"venue":null,"work_id":"bbd01b99-154a-496c-9dd9-93a0c2e61895","year":2014},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.035826Z"},"links":{"citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:f358a2fb6d6ee98b7d2a7f2b233185eaca92905af7d37adcf639bba6b3139a23","observation_id":"69fd76ab-d1c2-4bb5-b7d0-5fcd3a33e81c","resolution":{"observed_at":"2026-08-16T05:46:49.266577Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1505.04474","last_updated":"2015-05-17T23:21:35Z","snapshot_observed_at":"2026-08-19T22:00:06.074211Z","submitted_at":"2015-05-17T23:21:35Z","title":"Visual Semantic Role Labeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1505.04474","snapshot_observed_at":"2026-08-16T05:46:49.026653Z","title":"arXiv preprint arXiv:1505.04474","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.026653Z"},"links":{"cited_paper":"/paper/1505.04474","citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:84396480ff39464dd65fd9a0b7b695416f9c21a1536746df64fe97b18bedad8d","observation_id":"7d4d5ae1-453f-4e44-be16-9dbce1c2165b","resolution":{"observed_at":"2026-08-16T05:46:49.026653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:46:49.234429Z","title":null,"venue":null,"work_id":"83931c51-40c7-4f21-ba3c-c3c9f37a3e98","year":2016},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.060359Z"},"links":{"citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:2dcbc95fdfccd6d0d46f77142dcf9cdb48038920b8a99e69376ce4025805fd65","observation_id":"000c0eaf-9edc-41ad-b5fa-5f15b6197962","resolution":{"observed_at":"2026-08-16T05:46:49.239473Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:46:49.298544Z","title":null,"venue":null,"work_id":"7af860b5-af5b-40c2-b2c7-fa1564a06274","year":2018},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.001638Z"},"links":{"citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:90c4b32b492824ee68d7caa8934d1ea5e23e897ace592b821cd92625b66595b0","observation_id":"c0ccbfd0-cdea-4a78-8d0d-0737ae1c1618","resolution":{"observed_at":"2026-08-16T05:46:49.302854Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-16T09:25:53.087782Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-16T05:46:49.016559Z","title":"arXiv preprint arXiv:2010.11929","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.016559Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:1436f5b11e0db5680a82f9f73653ea3ee46779f21182455a9f6d7ffc63405d76","observation_id":"e0027bfc-2a46-4ce4-9c31-95518fe02a17","resolution":{"observed_at":"2026-08-16T05:46:49.016559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:46:49.249594Z","title":"Computational Intelligence and Neuroscience 2021, 9922697","venue":null,"work_id":"6c98f5cf-56e0-40c4-89f1-505abdd83e34","year":2021},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.050782Z"},"links":{"citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:71757ad50a88d8826e7a6855edeab51428305452d95050cec69c43497ef72b1f","observation_id":"9d98046a-a8ce-40f1-ba50-bbf7e160daae","resolution":{"observed_at":"2026-08-16T05:46:49.253759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.12329","last_updated":"2022-03-30T16:04:38Z","snapshot_observed_at":"2026-08-16T17:25:09.013446Z","submitted_at":"2022-01-28T18:51:09Z","title":"DAB-DETR: Dynamic Anchor Boxes are Better Queries for DETR","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.12329","snapshot_observed_at":"2026-08-16T05:46:49.040494Z","title":"arXiv preprint arXiv:2201.12329","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.040494Z"},"links":{"cited_paper":"/paper/2201.12329","citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:7bc703dd66910c0617153ae487d5b571dddf55f2893c20fc1603205bad03b65d","observation_id":"12c53fd1-4979-40be-8f94-cdf54f4238d7","resolution":{"observed_at":"2026-08-16T05:46:49.040494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-16T05:46:48.995926Z","title":"arXiv preprint arXiv:2303.08774","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:48.995926Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:1747b25e8d83b0304632d2ac24cfdf5b31e1d4c7f828638101cdcfa69ffcfa6b","observation_id":"5430cd55-03d5-49a6-9a97-f832eb95c26a","resolution":{"observed_at":"2026-08-16T05:46:48.995926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-16T05:46:49.064632Z","title":"arXiv preprint arXiv:2408.00714","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-16T05:46:49.064632Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2504.19847"},"observation_digest":"sha256:bb22828842e89cd241d588a619f2120cf15ff6991de658285864628acfe26d09","observation_id":"f6671e11-45bd-4de5-842c-84b6e5b99c4c","resolution":{"observed_at":"2026-08-16T05:46:49.064632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2504.19847","last_updated":"2025-04-28T14:45:26Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-19T21:14:44.191321Z","submitted_at":"2025-04-28T14:45:26Z","title":"Foundation Model-Driven Framework for Human-Object Interaction Prediction with Segmentation Mask Integration"},"reference_resolution":{"displayed":17,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":1},"total_outbound_references":17},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 17 of 17 outbound references and 0 inbound Pith citation observations for arXiv:2504.19847."}