{"as_of":"2026-08-06T18:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f72fe1089a9596164ba714c1c268f2f99c04d2ada80f8b6f8f4f72646f00dd64","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-26T08:50:10.781971Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.23256/citation-record","integrity":"/paper/2606.23256/integrity","json":"/paper/2606.23256/citation-record.json","paper":"/paper/2606.23256"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Vivit: A video vision transformer","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:a0064d002b1e22076ad615c9943a0be4d4d4cca830ce845c3a70fc6fb3a12e25","observation_id":"30762952-72a7-48bd-bc06-7c3a8b3e541d","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Hiervl: Learning hierarchical video-language embeddings","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:5e725b092762e9ce1292af040079b517040c8add4f95aebc7c7c1fa7a0d66668","observation_id":"06fe4d81-9776-4d8d-9179-cfaee721a028","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:4ef8e818edd34ca225716b707b4addd435061a719c4b1077334ef610c5bd7c78","observation_id":"eaf34c33-27d3-4183-832f-2c3259b86b69","resolution":{"observed_at":"2026-07-04T10:29:44.920591Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"How much temporal long-term context is needed for action segmentation? InProceedings of the IEEE/CVF International Conference on Computer Vision, pages 10351–10361, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:221ee9bc029340d42981ef46d98e837a2c0bac20fad59062a02778f52130f449","observation_id":"3da0e6e3-d6c1-474f-85d2-583c6f207800","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"My view is the best view: Procedure learning from egocentric videos","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:336d287e58719a41dc9f96f54d124282cab274085f802b130673daff1c9c248e","observation_id":"23d89db5-1e21-4915-aeb7-475487c61feb","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"United we stand, divided we fall: Unitygraph for unsupervised procedure learning from videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:978595af6ac5b09250ee4b15b02bc2d9ab0ac8df76ab10b9829b1783b000785c","observation_id":"b0c1c981-328a-4bf3-9c43-b96cdbc7e24b","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08471","last_updated":"2024-02-15T18:59:11Z","snapshot_observed_at":"2026-08-02T05:40:31.086014Z","submitted_at":"2024-02-15T18:59:11Z","title":"Revisiting Feature Prediction for Learning Visual Representations from Video","version":1},"cited_work":{"arxiv_id":"2404.08471","doi":"10.1145/3178876.3185996","metadata_source":"pith","pith_arxiv_id":"2404.08471","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Revisiting Feature Prediction for Learning Visual Representations from Video","venue":"cs.CV","work_id":"f7251dcf-5341-4915-bfe7-27812387b61a","year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/2404.08471","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:34766f8555fc015d03f3a6037b5de2a0b153b06ea2dc099a314b55e439340282","observation_id":"e8b411a7-43f5-43b1-9a15-60e3436c6fa6","resolution":{"observed_at":"2026-07-04T10:29:44.931885Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-23T01:54:34.489718+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T01:54:34.489718+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Unified fully and timestamp supervised temporal action segmentation via sequence to sequence translation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:5f24fe0e96b721d1520c98b371ecafa56aef6bf958a30a635286f5dda57f54e1","observation_id":"8b162020-9d2a-4d11-bcfd-b631d8e2a47f","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:132fd92662a5e6ff4f043cd3ce47da4f070345e29a4d00a0ec71482130cbf009","observation_id":"f6d52c73-000b-4da7-bb27-051020c1db05","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Quo vadis, action recognition? a new model and the kinetics dataset","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:632eaf14ff9609ac41887d88af006602d6f29f0d77fa6551ae3b9c58de08f0bc","observation_id":"7bbae96d-5dc0-4b67-acff-61f53883c12f","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Streaming videollms for real-time procedural video understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:c5c731dd07ea42bdf7347feae7ab31d16e92e2a64bb116411a203fded61f309c","observation_id":"88ba19b6-fd84-4cec-b005-5f0d3c09f131","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.10942","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T19:46:28.673420Z","title":"arXiv preprint arXiv:2512.10942 (2025)","venue":null,"work_id":"804987cf-5769-4ab2-9862-b70e398ff385","year":2025},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:e182fc4f2143df9962d3e552a29c5815f7bb3b4d610ea3fd5aea238503ca72d9","observation_id":"adfd7a1b-c00b-40d6-b48d-d34f0bb6a259","resolution":{"observed_at":"2026-07-04T10:29:44.907216Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Videollm-online: Online video large language model for streaming video","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:782231e0907250ea21fc5cf3987c565aab4c2872b5a4cbd49d72c5b6c5dfcc73","observation_id":"71aa14d4-081b-4f1c-aa51-6904dfc229d1","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:402a29cf21a9aceb8ae1f8e73043aa4ef9849034cdba594511a2c7a80f245a98","observation_id":"5364542d-b0f3-43cd-97e7-36697146a388","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Unsupervised procedure learning via joint dynamic summarization","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:0584a41e17b37ab3d1e4a689232bb51812287beb411f9f498b81a05e9cbeea19","observation_id":"00e27043-8aea-4f77-9b04-eed8cf749986","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Multiscale vision transformers","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:cedb4b302f3ba5ebfa55b7d7b9a6f1bccdf8cc3c8ff2b9bb9854586a19a8074c","observation_id":"df23605c-390b-4893-b90b-c47ee46ae557","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Ms-tcn: Multi-stage temporal convolutional network for action segmentation","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:d99d8ed9c6241de0b28f174f653ee56178bbf171c227e442659039430b2537a3","observation_id":"11e5cc62-821d-420f-b000-10e97d582ceb","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:25e352170e5e946470f6c3c3605c7952e02c0f1e457fc302996ea9e5e2d8aa82","observation_id":"3a446f45-2d4b-4628-a78d-1ca0422f4703","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Ego-exo4d: Understanding skilled human activity from first-and third-person perspectives","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:a14047e9f3ecea3e3a8e2c74b605faced889a6fa4d66593cdf6037417a43f662","observation_id":"1b034fb5-208c-4a7a-8c5f-5e15fc5b790e","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Ma-lmm: Memory-augmented large multimodal model for long-term video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:4aebc5d19d163842fda0133adbdabad26c0ddfd64ffa38bbaf2de511b836763a","observation_id":"d7018eb6-96e9-4cf5-bf6e-6fef59e1f2cc","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21055","last_updated":"2025-08-12T04:19:14Z","snapshot_observed_at":"2026-07-06T20:59:22.529105Z","submitted_at":"2025-03-27T00:03:55Z","title":"What Changed and What Could Have Changed? State-Change Counterfactuals for Procedure-Aware Video Representation Learning","version":6},"cited_work":{"arxiv_id":"2503.21055","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.21055","snapshot_observed_at":"2026-07-04T10:29:44.908509Z","title":"What changed and what could have changed? state-change counterfactuals for procedure- aware video representation learning.arXiv preprint arXiv:2503.21055","venue":null,"work_id":"56bf56a8-c9c0-40b0-a770-ab8ac6c8923b","year":2025},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/2503.21055","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:375cb0158c2ed716fb36f52b013e1515f8501db5a4ceb85347403ce38c5c9a0c","observation_id":"b061230b-40aa-434d-a92a-4c5602cf849b","resolution":{"observed_at":"2026-07-04T10:29:44.910139Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Video token merging for long video understanding.Advances in Neural Information Processing Systems, 37:13851–13871, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:d755ed617e651a380885c026fa3ee91a77a9c7955593f23828254b91b334f403","observation_id":"7d9afdd9-8058-4dfc-bf36-35fc5bb7caed","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06166","last_updated":"2024-10-08T16:10:29Z","snapshot_observed_at":"2026-08-03T18:47:03.750114Z","submitted_at":"2024-10-08T16:10:29Z","title":"Temporal Reasoning Transfer from Text to Video","version":1},"cited_work":{"arxiv_id":"2410.06166","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.06166","snapshot_observed_at":"2026-07-04T10:29:44.916128Z","title":"Tem- poral reasoning transfer from text to video","venue":null,"work_id":"6ed3ce19-4b99-4851-8f1c-ed95e050473b","year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/2410.06166","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:b2e37fd9dfdc791d2952173761ed1c133562546c3fc62050726e7a21dd44243b","observation_id":"0162f33f-b6be-4529-becb-9eabebc5d2fa","resolution":{"observed_at":"2026-07-04T10:29:44.917793Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2020.302175","doi":"10.1109/tpami.2020.3021756","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"IEEE Trans- actions on Pattern Analysis and Machine Intelligence pp","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","work_id":"c46f1335-3706-4ea6-9f20-9f0cc7c616ff","year":2020},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:f764cb861134ac2079dbae2da5375abcd207666ecad648ce4860475b08c8d370","observation_id":"5bbda93a-6220-42db-bad6-88ebc778d723","resolution":{"observed_at":"2026-06-26T08:59:15.355133Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-25T06:23:35.912285+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T06:23:35.912285+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Tsm: Temporal shift module for efficient video understanding","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:42f33c935fcdbe99a655124990f1d7291498cd46835b0521885244c832a9dfe5","observation_id":"f8fab75e-9cfe-40ad-9448-f9d310d6324c","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-04T21:17:37.211833Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"cited_work":{"arxiv_id":"2403.00476","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.00476","snapshot_observed_at":"2026-07-04T10:29:44.911015Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","venue":"cs.CV","work_id":"88f4d72e-ee93-420a-a61e-64b02b8cc454","year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/2403.00476","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:7cfbe8d140844a6e494b6935d173d947d1bde4c9af2d75613a5e8fa8d4310803","observation_id":"9f976101-5c0b-4d29-a98c-06ede3757fb5","resolution":{"observed_at":"2026-07-04T10:29:44.912158Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Video swin transformer","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:43477d45ccd0113b767705a49b9ccabd298cb246c6571a44cb5ea6b9c0e882e8","observation_id":"1b79aa2e-9f30-45db-92e4-78d5dd7a1f0b","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Fact: Frame-action cross-attention temporal modeling for efficient action segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:5c01240e0ac247d57a19ab3f775c9a1aa5ab856229ae334ccc7965464f719408","observation_id":"a546a6aa-27a6-4418-b6c9-32ae967c7b36","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Streamer: Streaming representation learning and event segmentation in a hierarchical manner.Advances in Neural Information Processing Systems, 36: 45694–45715, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:9dbb66978ed7172c40ee906f1475dac37bb70d3adf77164985cd8f5eeb8b1a89","observation_id":"fead2f64-dc68-4440-8ad1-adfa1389bf70","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.14482","last_updated":"2026-06-11T10:07:56Z","snapshot_observed_at":"2026-08-06T07:23:27.418432Z","submitted_at":"2026-03-15T17:02:40Z","title":"V-JEPA 2.1: Unlocking Dense Features in Video Self-Supervised Learning","version":3},"cited_work":{"arxiv_id":"2603.14482","doi":"10.48550/arxiv.2603.14482","metadata_source":"pith","pith_arxiv_id":"2603.14482","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2.1: Unlocking dense features in video self-supervised learning","venue":"cs.CV","work_id":"3da7cd7a-c3f3-4e48-9704-4320e54aec60","year":2026},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/2603.14482","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:018c257dfd8a2feefa5e15976d2242db4328d04eb646dffb7f9214245187c75d","observation_id":"9476243b-aa54-45e4-85f3-ef0d67dc355f","resolution":{"observed_at":"2026-07-04T10:29:44.926386Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Video transformer network","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:b48443a064021c7115456c680459f43a1e4b402098ff1e8e1cdb1528c667edbe","observation_id":"83678f8d-02ef-4142-8e04-514bee9b5d4d","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03748","last_updated":"2019-01-22T18:47:12Z","snapshot_observed_at":"2026-07-06T06:49:24.960992Z","submitted_at":"2018-07-10T16:52:11Z","title":"Representation Learning with Contrastive Predictive Coding","version":2},"cited_work":{"arxiv_id":"1807.03748","doi":"10.1609/aaai.v36i10.21390","metadata_source":"pith","pith_arxiv_id":"1807.03748","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Representation Learning with Contrastive Predictive Coding","venue":"cs.LG","work_id":"7b08a1d4-d565-424e-9c86-6ef244b7b90a","year":2018},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/1807.03748","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:74a89d022cb9a453697e024ac584d3411461a70d291af51856b8e248e49aa0c4","observation_id":"c11bbe3b-a71b-4b30-b643-19820b9c79e8","resolution":{"observed_at":"2026-07-04T10:29:44.912567Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Hiero: understanding the hierarchy of human behavior enhances reasoning on egocentric videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:c5b55792907cec2bb40d01354644a016546f69816ee04b77efb1f9d4b2bcda15","observation_id":"c51aedc1-5ff2-4ddc-bc71-e78bd51f5680","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Omnia de egotempo: Benchmarking temporal understanding of multi-modal llms in egocentric videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:736ff84554fa2935d9fce713e6dcbfe27928a2a409b888b2f783357ec37f0942","observation_id":"4ce97aa0-80c5-4523-80a0-8eaf7a7406b0","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Egovlpv2: Egocentric video-language pre-training with fusion in the backbone","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:2294a0c02545f571df6cb72cbbef91df4b3bc6ae890041a3772a670299d8acbc","observation_id":"cd01ca72-9e9d-4e0a-822e-3a5f0a10473f","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Learning from untrimmed videos: Self-supervised video representation learning with hierarchical consistency","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:ecf10fc672348b98d2b74b97d546a4f84f14ab7b3816c6257e1a16857b8d44f8","observation_id":"6bca6163-4a9e-44c0-9974-be842e71cff4","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16998","last_updated":"2025-06-11T17:46:56Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:59:09Z","title":"Understanding Long Videos with Multimodal Language Models","version":5},"cited_work":{"arxiv_id":"2403.16998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16998","snapshot_observed_at":"2026-07-04T10:29:44.913681Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":"a0d8f834-29b0-4597-ad47-382843695ca9","year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/2403.16998","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:dcf3de14a3da3e7ddb64c741b5043070e37a0f978aa41c9f136e18971c7e73ee","observation_id":"0a58456e-931f-4ce1-a837-dae08856644d","resolution":{"observed_at":"2026-07-04T10:29:44.915306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Timechat: A time-sensitive multimodal large language model for long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:53bfb41bee4495d14e68b4b9a71cc2695a1183050c0032ad4e67af8bbe716591","observation_id":"bc406de2-06fd-4ddc-bae5-ee79b19bb3d7","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Assembly101: A large-scale multi-view video dataset for understanding procedural activities","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:fe9369c5b3d379071e51d6413da75c545bb2a5d3b77b3f8508a2d0f3b6a96620","observation_id":"ef084bf9-244c-42c1-98ff-a33632dfc924","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Video-xl: Extra-long vision language model for hour-scale video understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:b6f677ee7011f376cc1774150991bc0c62dfffe5118ef67ce29c742949e0d6f4","observation_id":"3931da6f-60a2-49af-9c11-1c12b3e661a4","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"C2f-tcn: A framework for semi-and fully-supervised temporal action segmentation.IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(10): 11484–11501, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:a12fd911ca2a9e71981e8577cf98d38c4e167752b03a098bf1a5afd014683e8f","observation_id":"d22fa270-00bf-4df9-9cdb-f4902d7533be","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Moviechat: From dense token to sparse memory for long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:3e38bfdc9c044a6905fa77851a86d1351ceb621c51138383cc9b466113d29504","observation_id":"46b509a0-3dd0-476b-807d-d1fe345c9914","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Moviechat+: Question- aware sparse memory for long video question answering.IEEE Transactions on Pattern Analysis and Machine Intelligence, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:fdd285ddc0090434eaab1bdabfe2f21b06eea5b0859e59eccf986bf9cb66e547","observation_id":"123104cd-b2ce-440a-8f35-f9ac36f28727","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Roformer: Enhanced transformer with rotary position embedding.Neurocomputing, 568:127063, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:0767971c63248bfe74717681e1ba96f8a48ae876d26a24ba1b7cb2f7e07dd349","observation_id":"b2b77dd9-732d-48b1-90ab-5559d9311043","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Koala: Key frame-conditioned long video-llm","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:996a47d3828be3c3e34df7019725120d6f682f92fd1719c1a5ecc818a654b118","observation_id":"df19cd4c-1f83-4b27-97c6-3020bc27712c","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training.Advances in neural information processing systems, 35: 10078–10093, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:7bc4ffdd001e33699485cce2d5b91b0a00438a7354dd7766d846f40ed6391027","observation_id":"a501836b-12c1-4f49-8374-bbf318a41535","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":"2409.12191","doi":"10.48550/arxiv.2409.12191","metadata_source":"pith","pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","venue":"cs.CV","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:8b6056f9703ed5f1104e524aef6031abbd99afbe7d880cba80c98398e12796f9","observation_id":"a7d9282e-2770-4d41-8c3b-578a1d32c1c9","resolution":{"observed_at":"2026-07-04T10:29:44.917802Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding.Advances in Neural Information Processing Systems, 37: 28828–28857, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:9e9e6f46915722c0eecadecac050ee209cb7cb9025ea36a006b8cb79bd60bfcc","observation_id":"d6c02184-a138-4520-95a1-de0289740a77","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Videollm-mod: Efficient video-language streaming with mixture- of-depths vision computation.Advances in Neural Information Processing Systems, 37:109922–109947, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:1b93af74b037dac2aabdd2fa3ffb3797ad905beb4bb6c8d5d2a44dbed9bae7d3","observation_id":"0c7eca5d-c4c6-47f5-abb5-08d848f2ad82","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Hierarchical self-supervised representation learning for movie understanding","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:ce4362e7ab3359b2fb8791b4f33df7ca75e8a2d6972e0dcf23855f8b1c68332b","observation_id":"beb185d8-738a-44b3-a01f-5ef2ba2254cb","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.08568","last_updated":"2021-10-16T13:07:20Z","snapshot_observed_at":"2026-07-06T11:58:34.888164Z","submitted_at":"2021-10-16T13:07:20Z","title":"ASFormer: Transformer for Action Segmentation","version":1},"cited_work":{"arxiv_id":"2110.08568","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2110.08568","snapshot_observed_at":"2026-07-04T10:29:44.921879Z","title":"Asformer: Transformer for action segmentation","venue":null,"work_id":"b26159a6-7bdf-44eb-b63c-2f839236d2f9","year":2021},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"cited_paper":"/paper/2110.08568","citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:e90bb1f603cf197cc4ed915c71f9587202a76d0e45e162432da17ed1ff16c67f","observation_id":"4f24f1a1-63c1-4daf-b518-729a621678e3","resolution":{"observed_at":"2026-07-04T10:29:44.923552Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Learning procedure-aware video representation from instructional videos and their narrations","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:6e3d9f12bc44618babd6ccd3b7818ecb493e854395737a2fbc77c28a046b67eb","observation_id":"77c22baa-aadc-4027-8c66-000783c501fc","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T08:50:10.781971Z","title":"Procedure-aware pretraining for instructional video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:f0c0be15222ec2802b9662a6f226dcb7c394a9823f1171a52dfb0c20ce380458","observation_id":"b063b89b-3244-4784-9e16-80dc3c6b9334","resolution":{"observed_at":"2026-06-26T08:50:10.781971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"5890.0839","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T10:29:44.927964Z","title":null,"venue":null,"work_id":"c3343253-29ee-460d-96bd-2f13b5769b9a","year":2048},"citing_paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-26T08:50:10.781971Z"},"links":{"citing_paper":"/paper/2606.23256"},"observation_digest":"sha256:12d6ae1013e883a90ca950c5c62b0271a093f4763c70db95cadf45d650471be2","observation_id":"7d1a366b-694e-4372-a3c8-d2394234f837","resolution":{"observed_at":"2026-07-04T10:29:44.929335Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2606.23256","last_updated":"2026-06-22T12:38:36Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-05T15:59:57.396281Z","submitted_at":"2026-06-22T12:38:36Z","title":"P-JEPA: Procedural Video Representation Learning via Joint Embedding Predictive Architecture"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":41,"verified_exact":13,"verified_fuzzy":0},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 0 inbound Pith citation observations for arXiv:2606.23256."}