{"as_of":"2026-08-18T08:14:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2e60a5a38de0a3da9b99d665ee1aff366437bd713d531062d7e107228df61e99","coverage":[{"denominator":19,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T18:07:42.779884Z","state":"measured"},{"denominator":19,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":19,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.21161/citation-record","integrity":"/paper/2507.21161/integrity","json":"/paper/2507.21161/citation-record.json","paper":"/paper/2507.21161"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:43.213536Z","title":null,"venue":null,"work_id":"5cb67386-5a64-440a-ab58-b870bda98c87","year":2019},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.684756Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:cc03ed0351c536783e7a5ec89871f867d14b35b404ec47b9e4862be4f30c8771","observation_id":"9aa94afd-b4bb-41f9-a00a-b0d21603b093","resolution":{"observed_at":"2026-08-15T18:07:43.222436Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:43.186140Z","title":"Do they want to cross? understanding pedestrian intention for behavior prediction","venue":null,"work_id":"71d81aeb-fc3a-470c-bee2-b918d7ed7435","year":2020},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.690063Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:aa32a2aac6ca13cd1cf44efde210959fbd3be6374c4f4481aa07e359de6b971c","observation_id":"f71bc149-f084-4425-aa0f-4c289a64c7dc","resolution":{"observed_at":"2026-08-15T18:07:43.194599Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:43.159649Z","title":"Long-term on-board prediction of people in traffic scenes under uncertainty","venue":null,"work_id":"f0e9d18c-0f20-4941-848b-427091fca285","year":2018},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.695044Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:78fe81f6b608f22f50dd9b89e230d8a15f76ce7e96e9fdf107d989296e6144c1","observation_id":"2c1b9d46-d26d-4580-8608-ebde7765537c","resolution":{"observed_at":"2026-08-15T18:07:43.170587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.06582","last_updated":"2020-05-13T20:59:37Z","snapshot_observed_at":"2026-08-10T13:07:54.040841Z","submitted_at":"2020-05-13T20:59:37Z","title":"Pedestrian Action Anticipation using Contextual Feature Fusion in Stacked RNNs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.06582","snapshot_observed_at":"2026-08-15T18:07:42.699578Z","title":"Pedestrian action anticipation using contextual feature fusion in stacked rnns","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.699578Z"},"links":{"cited_paper":"/paper/2005.06582","citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:45e9853a88e4fc9ccf9a7e10838026173e8b470ce4decaba9f5aae9e304622cc","observation_id":"b9b77231-fafb-471f-98f2-5c240693665c","resolution":{"observed_at":"2026-08-15T18:07:42.699578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:43.136791Z","title":"Pedestrian graph +: A fast pedestrian crossing prediction model based on graph convolutional networks","venue":null,"work_id":"3cd9e30b-d56d-45f6-9bef-f9aa526f4f55","year":2022},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.704924Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:98b94a65f428fccbb37c5688e0551569fa58153a388b828943e57b681fccc6f8","observation_id":"aeca8c1e-a786-4016-b8ff-dd6a97a751c7","resolution":{"observed_at":"2026-08-15T18:07:43.143086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:43.114975Z","title":"St cross- ingpose: A spatial-temporal graph convolutional network for skeleton- based pedestrian crossing intention prediction","venue":null,"work_id":"17302065-06d8-48c8-89a2-8bcba227251e","year":2022},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.710647Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:692442308cd74e3baebea6ae695a9a23d177cdf3afee92449f7cff7f11c2dddb","observation_id":"311116e4-5ac7-40f2-abe1-d676710958e0","resolution":{"observed_at":"2026-08-15T18:07:43.122366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:43.096586Z","title":"Pit: Progressive interaction transformer for pedestrian crossing intention prediction","venue":null,"work_id":"75b38ba8-ac56-4341-aecb-e25f73caaa15","year":2023},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.715984Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:9b4bf66bd2a469abbaf3e013fde869f172ab37a4c8ba6fcd274ea564be47aa81","observation_id":"7eb3bf88-4791-473d-b572-37bd297b0ffd","resolution":{"observed_at":"2026-08-15T18:07:43.102444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.08647","last_updated":"2021-05-18T16:23:15Z","snapshot_observed_at":"2026-08-16T18:23:53.437132Z","submitted_at":"2021-05-18T16:23:15Z","title":"IntFormer: Predicting pedestrian intention with the aid of the Transformer architecture","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.08647","snapshot_observed_at":"2026-08-15T18:07:42.721236Z","title":"Intformer: Predicting pedestrian intention with the aid of the transformer architecture","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.721236Z"},"links":{"cited_paper":"/paper/2105.08647","citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:7afe4a76a7a3ffd45bdf4f6a3c71a0e76a793c4e8e94e9e2fd0307b741d4ed40","observation_id":"3a2202ff-0452-4d8d-a250-c6769b3f62b7","resolution":{"observed_at":"2026-08-15T18:07:42.721236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:43.075426Z","title":"Multi-input fusion for practical pedestrian intention prediction","venue":null,"work_id":"7075a1da-ec17-4d9a-ac69-9fddf1765d91","year":2021},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.726289Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:0ee5aac3eb1be28838d9bfbfcd850b93dec04a399a58f4d1995f96455774bff7","observation_id":"30013173-8e2d-4bd9-9052-761c46394a6d","resolution":{"observed_at":"2026-08-15T18:07:43.082875Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:43.055134Z","title":"Mcip: Multi-stream network for pedestrian crossing intention prediction","venue":null,"work_id":"c86c9a37-50a0-4c6a-a279-5695186583d3","year":2022},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.730697Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:e65665781b391d410dd377036e7a32667c588ed67c46645f2a9effe720223138","observation_id":"9a4b6d00-7ac2-4067-a77d-6181dbaf8c2b","resolution":{"observed_at":"2026-08-15T18:07:43.060484Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-15T18:07:42.735678Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.735678Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:bc9ee850759694446395e8958989f9723c36342111b1550cb1a8b79a02a19583","observation_id":"ad55b0f5-960b-4119-95e3-3b05f7e8a084","resolution":{"observed_at":"2026-08-15T18:07:42.735678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:42.741551Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.741551Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:d6222e771f9bf8bf5f8435cae605a8521936da80a886313c34a6650231daae15","observation_id":"56615001-b359-4991-8388-fa5ae81c44ea","resolution":{"observed_at":"2026-08-15T18:07:42.741551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-15T18:07:42.746267Z","title":"Llama: Open and efficient foundation language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.746267Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:4453c44ec1f354dabb2f91215533acde249de9672779de779913ad4074c1b6fa","observation_id":"06713f6c-bdea-4027-8be7-cbf2fef2493f","resolution":{"observed_at":"2026-08-15T18:07:42.746267Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:43.020793Z","title":null,"venue":null,"work_id":"ae1bd5e1-e2d1-4e99-bfa4-a5f794653ea9","year":2024},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.754285Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:160ad10b19dbebe64b6222734f634dc1a404f4ce949912f8164d759f45cde984","observation_id":"43118ef7-50b4-4739-b8ca-8ed0aa49908f","resolution":{"observed_at":"2026-08-15T18:07:43.028183Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.14786","last_updated":"2024-01-25T20:55:16Z","snapshot_observed_at":"2026-08-16T22:47:42.549190Z","submitted_at":"2023-11-24T18:02:49Z","title":"GPT-4V Takes the Wheel: Promises and Challenges for Pedestrian Behavior Prediction","version":2},"cited_work":{"arxiv_id":"2311.14786","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.14786","snapshot_observed_at":"2026-08-15T18:07:42.826910Z","title":"GPT-4V Takes the Wheel: Promises and Challenges for Pedestrian Behavior Prediction","venue":"cs.CV","work_id":"b10524a5-6b9c-47e4-a684-c9b41427b419","year":2023},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.759290Z"},"links":{"cited_paper":"/paper/2311.14786","citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:bc8fb50a29f94c53b0e7fc75f135682a71ebb3c1ac5bad9d1ade9e5026716626","observation_id":"d9e2c354-7181-4304-a8fd-b73f4f21bf68","resolution":{"observed_at":"2026-08-15T18:07:42.835595Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:42.995965Z","title":"Predicting pedestrian crossing intention with feature fusion and spatio-temporal attention","venue":null,"work_id":"496f4402-1da4-47a7-bfa9-7fd572b3265a","year":2022},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.765387Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:24447bcd2ea0a67f860bd4d84b4774e1b68f9b8cac92acbd9b4bebb33193aeed","observation_id":"8d40db08-b0ec-49b7-b8bf-882c2f8f91a9","resolution":{"observed_at":"2026-08-15T18:07:43.005860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:42.770137Z","title":"Benchmark for evaluating pedestrian action prediction","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.770137Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:ad932bc23e9a4ad39d6db7feb824faa6422eea0fa74931fdda7fa6e9717360a7","observation_id":"c78d25c9-d830-4a76-bf63-dc32538967d9","resolution":{"observed_at":"2026-08-15T18:07:42.770137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:42.960659Z","title":"”Role play with large language models.” Nature 623, no","venue":null,"work_id":"3be1e584-3599-4f61-bd33-cd47b598cb27","year":2023},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.775070Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:f04574120b9588d13e103d9854de4cdb75c2a9fb5e39d676c168bc1ef08ed209","observation_id":"b0427b0a-b3a7-4748-ac8c-3159ed965c5c","resolution":{"observed_at":"2026-08-15T18:07:42.965768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:07:42.943466Z","title":"”Towards revealing the mystery behind chain of thought: a theoretical perspective.” Advances in Neural Information Processing Systems 36 (2023): 70757-70798","venue":null,"work_id":"4fe1ecb5-2acb-4059-8f3f-2dfc49267209","year":2023},"citing_paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T18:07:42.779884Z"},"links":{"citing_paper":"/paper/2507.21161"},"observation_digest":"sha256:723264c8ef89c0b62b792e852aa2d1c56fd083d59965b5a1210db8366b7cf091","observation_id":"035527af-9384-47f2-bef9-3b18660b9c96","resolution":{"observed_at":"2026-08-15T18:07:42.949345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.21161","last_updated":"2025-07-25T07:23:11Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-15T19:29:16.287065Z","submitted_at":"2025-07-25T07:23:11Z","title":"Seeing Beyond Frames: Zero-Shot Pedestrian Intention Prediction with Raw Temporal Video and Multimodal Cues"},"reference_resolution":{"displayed":19,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":1,"verified_fuzzy":10},"total_outbound_references":19},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 19 of 19 outbound references and 0 inbound Pith citation observations for arXiv:2507.21161."}