{"as_of":"2026-08-11T17:03:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:246e641b4131560c32153545d48d7ea581849fb37c700a14fc501d0fdc742ea8","coverage":[{"denominator":95,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":95,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T20:33:54.320561Z","state":"measured"},{"denominator":99,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":99,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:24:36.541729Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-23T06:02:37.644844Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"cited_work":{"arxiv_id":"2501.07978","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.07978","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fa- cial dynamics in video: Instruction tuning for improved fa- cial expression perception and contextual awareness","venue":null,"work_id":"f0aef9b3-ec8d-4b60-981d-68951a751ecc","year":2025},"citing_paper":{"arxiv_id":"2501.05067","last_updated":"2026-04-20T07:42:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-09T08:43:57Z","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","version":3},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-05-23T06:01:00.775721Z"},"links":{"cited_paper":"/paper/2501.07978","citing_paper":"/paper/2501.05067"},"observation_digest":"sha256:d14c1bc9a96b0fbe32e59f2ea0c52712177533dc422ea6264cc59c23b98612d2","observation_id":"dfbf3f28-6334-4a74-834c-57a2e45d06e1","resolution":{"observed_at":"2026-05-23T06:02:37.649306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"cited_work":{"arxiv_id":"2501.07978","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.07978","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fa- cial dynamics in video: Instruction tuning for improved fa- cial expression perception and contextual awareness","venue":null,"work_id":"f0aef9b3-ec8d-4b60-981d-68951a751ecc","year":2025},"citing_paper":{"arxiv_id":"2503.09158","last_updated":"2026-05-09T06:28:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-12T08:33:46Z","title":"FaVChat: Hierarchical Prompt-Query Guided Facial Video Understanding with Data-Efficient GRPO","version":6},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-23T00:21:51.621582Z"},"links":{"cited_paper":"/paper/2501.07978","citing_paper":"/paper/2503.09158"},"observation_digest":"sha256:1596cb3d2a80e4fd39ac2a857c95c07e6368520ac259edd72763f7ad9e3e3fb2","observation_id":"9ba63ce5-3dbf-4d8c-90ca-ef657731fb73","resolution":{"observed_at":"2026-05-23T00:22:18.796437Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.07978","snapshot_observed_at":"2026-08-06T22:24:36.541729Z","title":"Facial dynamics in video: Instruction tuning for improved facial expression perception and contextual awareness.arXiv preprint arXiv:2501.07978, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.21862","last_updated":"2025-06-27T02:29:58Z","snapshot_observed_at":"2026-08-09T10:29:21.701927Z","submitted_at":"2025-06-27T02:29:58Z","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","version":1},"reference_index":108,"source":"pdf_text","source_observed_at":"2026-08-06T22:24:36.541729Z"},"links":{"cited_paper":"/paper/2501.07978","citing_paper":"/paper/2506.21862"},"observation_digest":"sha256:51d803bdf367a0c4850e9c79e0ff12b7b92943a79b6c484fb238defb60800b77","observation_id":"2f1f05b9-e8db-4b09-a634-0d7783dd5351","resolution":{"observed_at":"2026-08-06T22:24:36.541729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"cited_work":{"arxiv_id":"2501.07978","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.07978","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fa- cial dynamics in video: Instruction tuning for improved fa- cial expression perception and contextual awareness","venue":null,"work_id":"f0aef9b3-ec8d-4b60-981d-68951a751ecc","year":2025},"citing_paper":{"arxiv_id":"2605.18018","last_updated":"2026-05-18T08:09:37Z","snapshot_observed_at":"2026-07-06T23:28:55.079333Z","submitted_at":"2026-05-18T08:09:37Z","title":"See What I Mean: Aligning Vision and Language Representations for Video Fine-grained Object Understanding","version":1},"reference_index":107,"source":"pdf_text","source_observed_at":"2026-05-20T12:10:54.874012Z"},"links":{"cited_paper":"/paper/2501.07978","citing_paper":"/paper/2605.18018"},"observation_digest":"sha256:fd980c0c1a3d7677611fae7de56df7adcd290d23e38191db111f5ad23ec0e8f3","observation_id":"260b9555-6759-4738-8c89-c4f26d70f407","resolution":{"observed_at":"2026-05-20T12:13:16.336074Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2501.07978/citation-record","integrity":"/paper/2501.07978/integrity","json":"/paper/2501.07978/citation-record.json","paper":"/paper/2501.07978"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.884278Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.884278Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:1b5e8ac96cfb80724b1f5f74b5f53c1a9d7d822faae74c061c2b983d5b7ded9e","observation_id":"cc93ae32-cc76-4acc-ba60-cbec4923ba3e","resolution":{"observed_at":"2026-08-10T20:33:53.884278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.889385Z","title":"Emotion recognition in speech using cross- modal transfer in the wild, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.889385Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c090d7fc7c90fce152e764fd32cfc0525e46feaeff2d776aab43cae159f63539","observation_id":"4abc8b02-edbf-4fe1-a5c3-b75fdb296c4a","resolution":{"observed_at":"2026-08-10T20:33:53.889385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.893802Z","title":"Claude-3.5, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.893802Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:134d5276786608357a258c1c50b214cfa39443d8e6a0760b4c491bc7e8c4a082","observation_id":"cdd82447-b527-47b6-8bcc-f1a8e953315f","resolution":{"observed_at":"2026-08-10T20:33:53.893802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-10T20:33:53.903232Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.903232Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e03dc7724176fec4d7932448f119fccb929f01f2a72da281785203d5623ba226","observation_id":"d6f6185c-f02d-4cf3-a693-a66927114450","resolution":{"observed_at":"2026-08-10T20:33:53.903232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.907800Z","title":"Collecting highly paral- lel data for paraphrase evaluation","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.907800Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:fcf925298d0a07f9a80a2b1d7b1347227659b5bc8d73ea466c71feb5bc83cdb0","observation_id":"8cbe17d3-6f5a-421c-ba14-585c11660183","resolution":{"observed_at":"2026-08-10T20:33:53.907800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.02157","last_updated":"2025-06-24T07:42:09Z","snapshot_observed_at":"2026-08-11T11:10:36.034160Z","submitted_at":"2024-07-02T10:55:43Z","title":"FineCLIPER: Multi-modal Fine-grained CLIP for Dynamic Facial Expression Recognition with AdaptERs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.02157","snapshot_observed_at":"2026-08-10T20:33:53.913411Z","title":"Finecliper: Multi-modal fine-grained clip for dynamic facial expression recognition with adapters","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.913411Z"},"links":{"cited_paper":"/paper/2407.02157","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:6320af4e793046ee0d11e24aac812ab147f5beb939f59db521774a0fcdc9ed82","observation_id":"14417cbe-13db-43dc-8f44-377172cdc49b","resolution":{"observed_at":"2026-08-10T20:33:53.913411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04325","last_updated":"2024-06-06T17:58:54Z","snapshot_observed_at":"2026-07-06T18:26:42.010940Z","submitted_at":"2024-06-06T17:58:54Z","title":"ShareGPT4Video: Improving Video Understanding and Generation with Better Captions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04325","snapshot_observed_at":"2026-08-10T20:33:53.918678Z","title":"Sharegpt4video: Improving video understand- ing and generation with better captions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.918678Z"},"links":{"cited_paper":"/paper/2406.04325","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:930bea96e5ab951568372154b32b5a05b5d2a3af5ccd6b61a35e5aaa3993aa7a","observation_id":"7ade2e87-b39e-4fb9-a3b9-2a3bacd370d0","resolution":{"observed_at":"2026-08-10T20:33:53.918678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.923207Z","title":"Stcam: Spatial-temporal and channel attention module for dynamic facial expression recognition","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.923207Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:37409349e3315727b5d834d38ed960e98489dc6fbe07c29241fa4410773b98ff","observation_id":"a70c08be-1453-477b-a074-daa4cfb9a060","resolution":{"observed_at":"2026-08-10T20:33:53.923207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.14238","last_updated":"2024-01-15T15:23:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-21T18:59:31Z","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.14238","snapshot_observed_at":"2026-08-10T20:33:53.927345Z","title":"Internvl: Scaling up vision foundation mod- els and aligning for generic visual-linguistic tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.927345Z"},"links":{"cited_paper":"/paper/2312.14238","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:59e09778b724397b9c60ac9345c63f0e8dba3813f7cb828e3d34c57e5b210ab2","observation_id":"058f14fb-70b5-4ece-a08c-26afc70ea1c0","resolution":{"observed_at":"2026-08-10T20:33:53.927345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-10T20:33:53.930933Z","title":"Videollama 2: Advancing spatial-temporal modeling and audio understanding in video- llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.930933Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:bf0aa00f67ce04ffae48be612468264cf02e9dd537ecea7fa7bbc78292fd008b","observation_id":"8d653a64-b7f1-4921-926f-8b741118b8ba","resolution":{"observed_at":"2026-08-10T20:33:53.930933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.934690Z","title":"Transface: Calibrating trans- former training for face recognition from a data-centric per- spective, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.934690Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:716f72209ca380d7cf490b2229df0ba8f8870d12463ff8ddbb405f681102c0c4","observation_id":"184a51db-02cc-49ff-a15b-11671cca3d01","resolution":{"observed_at":"2026-08-10T20:33:53.934690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.937932Z","title":"Diffusionrig: Learning personal- ized priors for facial appearance editing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.937932Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:9a5ed9af30a9d8772f88053d0f646ad42e87cda38873b788cada59d9bbdc4c4b","observation_id":"70be547e-fdad-44e1-8463-20e460e9a722","resolution":{"observed_at":"2026-08-10T20:33:53.937932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16420","last_updated":"2024-01-29T18:59:02Z","snapshot_observed_at":"2026-08-05T03:42:54.599829Z","submitted_at":"2024-01-29T18:59:02Z","title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.16420","snapshot_observed_at":"2026-08-10T20:33:53.942983Z","title":"Internlm-xcomposer2: Mastering free-form text-image composition and compre- hension in vision-language large model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.942983Z"},"links":{"cited_paper":"/paper/2401.16420","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:dc2b1432ff8f97c11b10e4d3b1002fc05a8cd83ce5590d74cb5dfac137e57c70","observation_id":"868dbd0f-f384-43a1-8d93-af5d185986ac","resolution":{"observed_at":"2026-08-10T20:33:53.942983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.947008Z","title":"Strongsort: Make deep- sort great again","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.947008Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:8f1cd04428ebac1c27e7e35ce5d5408c644de868c2347d967adc0f3935346b62","observation_id":"6236c81b-5727-4fb4-8b60-8bc7774588a9","resolution":{"observed_at":"2026-08-10T20:33:53.947008Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.950534Z","title":"EmoCLIP: A Vision-Language Method for Zero-Shot Video Facial Ex- pression Recognition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.950534Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:631ed04507a74834e1ca743ecf1ff67ae268aff68acd4732ca8b374bba4be2df","observation_id":"17472c72-cec0-40a2-99b2-84c691b507e7","resolution":{"observed_at":"2026-08-10T20:33:53.950534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21075","last_updated":"2025-05-30T13:08:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:59:47Z","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21075","snapshot_observed_at":"2026-08-10T20:33:53.954454Z","title":"Video-mme: The first-ever compre- hensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.954454Z"},"links":{"cited_paper":"/paper/2405.21075","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:34496ea5a0f8359463c45a320e2d3ee8bf6b61ffa7afd9dbb200eee4984486ba","observation_id":"8e20eb0f-a12a-465a-9847-5d95dbc2c3b4","resolution":{"observed_at":"2026-08-10T20:33:53.954454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.15010","last_updated":"2023-04-28T17:59:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-28T17:59:25Z","title":"LLaMA-Adapter V2: Parameter-Efficient Visual Instruction Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.15010","snapshot_observed_at":"2026-08-10T20:33:53.958465Z","title":"Llama-adapter v2: Parameter-efficient vi- sual instruction model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.958465Z"},"links":{"cited_paper":"/paper/2304.15010","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:8c76eddac5fb69d4e0225ab7ca46f8050afac127985ec3a8afd2b8b392a1e77d","observation_id":"2ee3927a-96fb-42f8-b2fd-123ecbc2904c","resolution":{"observed_at":"2026-08-10T20:33:53.958465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.963217Z","title":"Music Emotion Recognition: Toward new, robust standards in personalized and context-sensitive ap- plications","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.963217Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:52ee2e061d69e18f16cbceb5bbf814f6a7b27e94e77d95ca82b61039584b949e","observation_id":"289d0ca9-4987-45a4-9795-01f08b933ada","resolution":{"observed_at":"2026-08-10T20:33:53.963217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.967225Z","title":"LoRA: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.967225Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:1be8e71d64b577adb41999354e6583b3308341ed0214c07643f3c87efc0bfa92","observation_id":"b69d2990-0a6e-4675-a16e-24c27a8da406","resolution":{"observed_at":"2026-08-10T20:33:53.967225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2011.11760","last_updated":"2020-11-10T21:49:14Z","snapshot_observed_at":"2026-08-06T11:42:54.284269Z","submitted_at":"2020-11-10T21:49:14Z","title":"Multimodal Pretraining for Dense Video Captioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2011.11760","snapshot_observed_at":"2026-08-10T20:33:53.972371Z","title":"Multimodal pretraining for dense video cap- tioning","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.972371Z"},"links":{"cited_paper":"/paper/2011.11760","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:f4451e24f871dbff9bcd195c4fe8d558fbfc73a48da6f76d15df4ad61b869700","observation_id":"b7602365-cb26-4dfd-ba34-4fa8ff056ba9","resolution":{"observed_at":"2026-08-10T20:33:53.972371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13250","last_updated":"2024-05-16T12:23:05Z","snapshot_observed_at":"2026-07-06T17:33:00.716054Z","submitted_at":"2024-02-20T18:58:54Z","title":"Video ReCap: Recursive Captioning of Hour-Long Videos","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13250","snapshot_observed_at":"2026-08-10T20:33:53.977080Z","title":"Video recap: Recursive captioning of hour-long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.977080Z"},"links":{"cited_paper":"/paper/2402.13250","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:a452b6ca93097008e29e768a71321df04e867f8a3aa9e7ee5e077af800e0eb84","observation_id":"4d501ff5-cfb1-45a9-8ef6-2707817343ae","resolution":{"observed_at":"2026-08-10T20:33:53.977080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.982395Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.982395Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e0f9648196c13425bcf7fb8d2b415b29449713f18d4c2d92122ac30230a685ae","observation_id":"33d0f7e5-2d02-4c8f-9663-7bed3fa6c57e","resolution":{"observed_at":"2026-08-10T20:33:53.982395Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.986831Z","title":"Dfew: A large-scale database for recognizing dynamic facial expres- sions in the wild, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.986831Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:7458e0e98803a3ccd0150222e801f1357f485f534b0ce6e9ba2514352602cf03","observation_id":"0d82b7f6-3584-472c-9619-29f1ddd9bf3a","resolution":{"observed_at":"2026-08-10T20:33:53.986831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.08046","last_updated":"2024-04-05T15:21:09Z","snapshot_observed_at":"2026-07-06T16:47:17.136168Z","submitted_at":"2023-11-14T10:11:36Z","title":"Chat-UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.08046","snapshot_observed_at":"2026-08-10T20:33:53.991893Z","title":"Chat-univi: Unified visual representation em- 9 powers large language models with image and video under- standing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.991893Z"},"links":{"cited_paper":"/paper/2311.08046","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c3bd165a526ec983989f10dd546157265cb89598ca73ffc632b0269efa49c90b","observation_id":"ac506810-02a2-41f5-af59-04498bbaecb6","resolution":{"observed_at":"2026-08-10T20:33:53.991893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.996900Z","title":"Expression, affect, action unit recognition: Aff-wild2, multi-task learning and arcface, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.996900Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:b9512946674ea7ebb4f4722434347f29994d320d6e7bb2e4cac4728c12e90484","observation_id":"b19a64f9-dbfe-445d-8e6b-174955e70bb1","resolution":{"observed_at":"2026-08-10T20:33:53.996900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.000691Z","title":"Afew-va database for valence and arousal estimation in-the-wild","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.000691Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e88c3b9fdceec2275f583834b4d1c669956c31f7677341e558c28eeb07fc0947","observation_id":"49941daa-11fa-4191-86d9-ffe12cf973be","resolution":{"observed_at":"2026-08-10T20:33:54.000691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.005726Z","title":"Dense-captioning events in videos","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.005726Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:ff06bfde6de64df8a154ebc2f93d468a2ca70b6a2dbd3aae55816f64188a4559","observation_id":"61377989-453c-4780-bcc3-f74d63bdc9c6","resolution":{"observed_at":"2026-08-10T20:33:54.005726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.426199Z","title":"Context-aware emotion recognition net- works","venue":null,"work_id":"42ffb5a3-fcda-42b7-b7f6-c3788f1098e0","year":2019},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.016846Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:ab65b2ca8a9a2550e3c91e8fdffdc49e6d8aad2e64fcd23bdfad3d372f8c0765","observation_id":"b64a0f55-59a7-4ea3-b690-8dfe4fe2cbb8","resolution":{"observed_at":"2026-08-10T20:33:55.430420Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.411498Z","title":"Llava-next: What else influences visual instruction tun- ing beyond data?, 2024","venue":null,"work_id":"213cd81d-41d5-421a-a420-1da739b20a2d","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.021962Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:3cf2cfac9415a171ef96e2d109f20eac2644b39178ff1f0600970962d153310e","observation_id":"c2632afb-5b1a-4883-b728-12af4c6b4f4e","resolution":{"observed_at":"2026-08-10T20:33:55.416512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-10T20:33:54.026558Z","title":"Llava-onevision: Easy visual task transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.026558Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:0ceca0198dfda741f4cce57a20bc61493475590abeeee16b07aef489dd19705c","observation_id":"7cff97f2-83c7-460b-8ba4-38f3988fdfee","resolution":{"observed_at":"2026-08-10T20:33:54.026558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07895","last_updated":"2024-07-28T19:58:08Z","snapshot_observed_at":"2026-07-06T18:44:24.873040Z","submitted_at":"2024-07-10T17:59:43Z","title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.07895","snapshot_observed_at":"2026-08-10T20:33:54.031175Z","title":"Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.031175Z"},"links":{"cited_paper":"/paper/2407.07895","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c9a8b7c90ea35e5fbe3e7002ab8b20df8a3ed34ad890d7f47d0eeeb0a4144e79","observation_id":"e26760f4-3fcf-4b4e-ac56-e16b76841d14","resolution":{"observed_at":"2026-08-10T20:33:54.031175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.035535Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.035535Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:200daccaafd19d56c2131e6816fd7cf8d60ff79095976cbb6ee1f2dde0e13c38","observation_id":"7f8c1543-a945-4910-af7e-409144b7f88f","resolution":{"observed_at":"2026-08-10T20:33:54.035535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T20:33:54.040489Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.040489Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:2ad2105cf09dfcc74b72094883744140f9442ecede506774cb5d94ae61aef181","observation_id":"2e33504e-cdb5-4ea3-9099-cc2a3bfd6e84","resolution":{"observed_at":"2026-08-10T20:33:54.040489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.391841Z","title":"Dual-sti: Dual-path spatial-temporal interac- tion learning for dynamic facial expression recognition","venue":null,"work_id":"4380bc23-76d7-45a2-89b3-eeb215afc379","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.045144Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:f9b23f1a638cd0d5c97731f66f38f678dcbea588010bbd83639e625dbfdc0a36","observation_id":"9094410c-9abc-4c79-bf0a-6f01cc769b6e","resolution":{"observed_at":"2026-08-10T20:33:55.395528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.380721Z","title":"Facial affective behavior analysis with instruction tuning, 2024","venue":null,"work_id":"dfffeb26-277d-4c0d-b6fa-30315ced2bed","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.049977Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:bd6f4d44c3054d1b0c1b3a19435e006c9e579769ca9f3df587a9005864ce2d37","observation_id":"b5f712da-2f25-4616-844c-628f256cccd3","resolution":{"observed_at":"2026-08-10T20:33:55.384319Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.368796Z","title":"Llama-vid: An image is worth 2 tokens in large language models","venue":null,"work_id":"d3932651-cd36-4165-a6e3-83d36968931f","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.054027Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:dedd9f78b49411342dbe2b77a97cd30f1aad0a705e828b803b24b8a1c587de9a","observation_id":"88837485-f367-4190-a78b-1cc280f44245","resolution":{"observed_at":"2026-08-10T20:33:55.373429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.352855Z","title":"Photomaker: Customizing realistic human photos via stacked id embedding","venue":null,"work_id":"b8a1b969-8583-49b6-aabb-c30b5f2cd349","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.058317Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:235ca2ca038766a41206fcf8e45be4a471c5dca73aea8d9ab10edd0e91372f47","observation_id":"231af9a6-5224-4f1f-9aa0-310260037293","resolution":{"observed_at":"2026-08-10T20:33:55.358412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10122","snapshot_observed_at":"2026-08-10T20:33:54.063000Z","title":"Video-llava: Learning united visual represen- tation by alignment before projection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.063000Z"},"links":{"cited_paper":"/paper/2311.10122","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:8d1eede3a323e23e373ff8cfdd4894f4dfe38efc2df4d62279d20b7a338b2bed","observation_id":"fa8073b5-3bd1-43cd-963c-2284534b242b","resolution":{"observed_at":"2026-08-10T20:33:54.063000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.337580Z","title":"Saanet: Siamese action-units attention network for improving dynamic facial expression recogni- tion","venue":null,"work_id":"09e02673-8a28-4223-b1db-e8ce0405cefb","year":2020},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.068366Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:a7bf394fab474e0c603144c92900b2ba282321eee1973780a1ed6300955b6992","observation_id":"c2d2eb12-c3d7-4f98-85e0-64962d28268b","resolution":{"observed_at":"2026-08-10T20:33:55.342396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03744","last_updated":"2024-05-15T19:22:44Z","snapshot_observed_at":"2026-07-06T16:28:22.350574Z","submitted_at":"2023-10-05T17:59:56Z","title":"Improved Baselines with Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03744","snapshot_observed_at":"2026-08-10T20:33:54.072952Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.072952Z"},"links":{"cited_paper":"/paper/2310.03744","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:2dee8309168ab3a050a3b1849ccd8fe0860c2a97581b537274614449aada8d36","observation_id":"b1ae5ed9-7d2a-466e-baf4-923d799a0625","resolution":{"observed_at":"2026-08-10T20:33:54.072952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.079056Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.079056Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:21de55ea435c4d84c0eff6f4a4dd697aeed570c5be8aa57a947285e8dfbe092c","observation_id":"deaaf7f4-3e1b-4922-a469-7e7c51f095d5","resolution":{"observed_at":"2026-08-10T20:33:54.079056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08268","last_updated":"2025-02-03T21:47:31Z","snapshot_observed_at":"2026-08-11T03:45:32.889856Z","submitted_at":"2024-02-13T07:47:36Z","title":"World Model on Million-Length Video And Language With Blockwise RingAttention","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08268","snapshot_observed_at":"2026-08-10T20:33:54.083552Z","title":"World model on million-length video and language with ringattention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.083552Z"},"links":{"cited_paper":"/paper/2402.08268","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:96f68dbd11288b1b63ad84f0209a153a8019f4b382d21a919059f723d520681e","observation_id":"2a8b8f70-e38a-45c8-8359-777e250eb065","resolution":{"observed_at":"2026-08-10T20:33:54.083552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.15785","last_updated":"2024-06-27T12:05:48Z","snapshot_observed_at":"2026-08-03T13:55:11.770435Z","submitted_at":"2023-09-27T16:58:35Z","title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.15785","snapshot_observed_at":"2026-08-10T20:33:54.087918Z","title":"One for all: Video conversation is fea- sible without video instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.087918Z"},"links":{"cited_paper":"/paper/2309.15785","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:6b0464b4e11aab1dd722fe5f791c9b82d48f78e2fb10e9941a50bd0556fa3c4d","observation_id":"c6196b34-0323-4b0c-a59a-c4796f0303b6","resolution":{"observed_at":"2026-08-10T20:33:54.087918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.00308","last_updated":"2024-03-30T10:11:26Z","snapshot_observed_at":"2026-08-10T01:25:17.951254Z","submitted_at":"2024-03-30T10:11:26Z","title":"ST-LLM: Large Language Models Are Effective Temporal Learners","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.00308","snapshot_observed_at":"2026-08-10T20:33:54.092524Z","title":"St-llm: Large language models are effective tem- poral learners","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.092524Z"},"links":{"cited_paper":"/paper/2404.00308","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c077b98887bb43563b29a6e5d336984ae436c0bb66fc0d7a6e790bcfe0ff0c75","observation_id":"4d075e73-3a34-4aa7-874c-120ee1be9681","resolution":{"observed_at":"2026-08-10T20:33:54.092524Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.317001Z","title":"Mafw: A large-scale, multi-modal, compound affective database for dynamic facial expression recognition in the wild, 2023","venue":null,"work_id":"09685e6b-d522-4d3f-893e-6a4c2e4c8546","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.098492Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:9372778fdbc4165c166e54486ebead7737005dd9da8020699e397fa2a27cfaea","observation_id":"803469f6-deef-4297-b87b-92833edf75ad","resolution":{"observed_at":"2026-08-10T20:33:55.321731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.305593Z","title":"DamoFD: Digging into backbone de- sign on face detection","venue":null,"work_id":"37a3b952-f17f-4090-8baa-ef331d1c52b9","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.102402Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:19802118dc5734c1f670dfc90c6b591674247d59a0eb3248a4c1d85a28ef0741","observation_id":"e8f43fe3-bbf1-4cc1-bce9-5e410716a276","resolution":{"observed_at":"2026-08-10T20:33:55.309781Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.294021Z","title":"Livingstone and Frank A","venue":null,"work_id":"05d57c00-64d1-4529-800b-a3ee5db3f76e","year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.109723Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:0b8de5b17b962a9989cc036c4af6b884a7fa54a89855255cd9d1b872d89aa267","observation_id":"1a21361f-515c-4376-a7f1-c07d45c73d82","resolution":{"observed_at":"2026-08-10T20:33:55.297610Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.281581Z","title":"Cohn, Takeo Kanade, Jason Saragih, Zara Ambadar, and Iain Matthews","venue":null,"work_id":"27715c5a-109f-4d61-8c27-2aadaa705023","year":2010},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.114235Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:8eb817f93c40d0ca898f68e5402853f6dcb23b27f180e15a85a1ef3329432186","observation_id":"c9ccb828-d05c-4b47-ae5e-5b61b57bade5","resolution":{"observed_at":"2026-08-10T20:33:55.286051Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.268278Z","title":"The extended cohn- kanade dataset (ck+): A complete dataset for action unit and emotion-specified expression","venue":null,"work_id":"4cc08af6-88b2-4234-98eb-dead70d2e3b4","year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.118502Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:13a4f1cb88ee2555f12408a60965b181c288f238a276cdd208e2ccbc43cd3407","observation_id":"2d697a4f-8b27-4320-b587-e8742ea614d5","resolution":{"observed_at":"2026-08-10T20:33:55.273230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.253957Z","title":"Learning multi-dimensional edge feature- based au relation graph for facial action unit recognition","venue":null,"work_id":"64f61333-c68f-4a5b-84db-f698d90b2305","year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.123016Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c5c8096d5c1c2aa80b3aa281268f304e0bbd9a767052ba73655057673ad1d110","observation_id":"b5c470e0-abcf-40aa-ba94-8c820facbbd1","resolution":{"observed_at":"2026-08-10T20:33:55.257974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-10T20:33:54.127444Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.127444Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:86b102bc8c00c4a0dd54a67d0a05ec4efed26846421f890938113e74a908aaa4","observation_id":"d0e643a4-7903-4f7c-91ee-b39f6e6d84e9","resolution":{"observed_at":"2026-08-10T20:33:54.127444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.242003Z","title":"Video-chatgpt: Towards detailed video 10 understanding via large vision and language models, 2024","venue":null,"work_id":"117f505e-6b26-4217-a4ae-09a3c7b73769","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.131730Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:7d17fa6f673a5712b862134946b09597784a80bb3245b622a3630c54edab5f9d","observation_id":"8ffd7c41-e2d5-4641-a30c-47452385392f","resolution":{"observed_at":"2026-08-10T20:33:55.246121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.135546Z","title":"Egoschema: A diagnostic benchmark for very long- form video language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.135546Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:229fa1dfcb1addf74db0652028b77bb7456cc9f5e5984c2b03859318f7eceea5","observation_id":"e82a982a-ec88-45ab-9a48-d0a347527cc1","resolution":{"observed_at":"2026-08-10T20:33:54.135546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.222997Z","title":"The importance of emotional regulation in mental health","venue":null,"work_id":"6132470b-0272-42fc-a6cf-9e10333cf85b","year":2022},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.139183Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:2fcbb400945f22d2e570374d6b6ab24bacca99d5e81482bf849ef02590a7a6d7","observation_id":"710122ae-8fba-4ba8-aa6f-3f5913cf3f7b","resolution":{"observed_at":"2026-08-10T20:33:55.227238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12960","last_updated":"2025-03-10T17:08:19Z","snapshot_observed_at":"2026-08-11T15:13:49.014950Z","submitted_at":"2024-03-19T17:58:04Z","title":"FaceXFormer: A Unified Transformer for Facial Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12960","snapshot_observed_at":"2026-08-10T20:33:54.143445Z","title":"Facexformer: A unified transformer for fa- cial analysis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.143445Z"},"links":{"cited_paper":"/paper/2403.12960","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:76294835462c17f692a146a88769c558ea8a016aa2598eea620109b1ef1bcdfe","observation_id":"719aa5fe-2e0a-4e13-9845-7ea75642b62f","resolution":{"observed_at":"2026-08-10T20:33:54.143445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.211652Z","title":"Repre- sentation learning and identity adversarial training for facial behavior understanding, 2024","venue":null,"work_id":"c765be70-4c0a-4ec5-97ac-28955d1ba679","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.148000Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:25ed68c512930741653649f0ec31f190dd5017b276f372d6a48ef7353c3f1e96","observation_id":"9f37c381-ba5c-4eb6-8586-9ccfaf984da2","resolution":{"observed_at":"2026-08-10T20:33:55.215683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.198643Z","title":null,"venue":null,"work_id":"e3785db6-fed0-41b0-8262-81bb22984a74","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.152472Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:8a41ca54020bf0530456c82a4d63b1cb10d5594133f21f6f60e7392a2c7d560a","observation_id":"d662422b-4de8-4bca-bc29-6399216d3fe9","resolution":{"observed_at":"2026-08-10T20:33:55.202907Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.185831Z","title":"Gpt-4v(ision) system card","venue":null,"work_id":"20f6d859-4610-495f-9930-bec31e7468f1","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.156778Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:b6042ff2561ec0796efe6a0609dadde0444f943ae1769b0116b925b513ce8f4b","observation_id":"f6f9678f-c6e2-4ba8-b2e7-9df8a433944e","resolution":{"observed_at":"2026-08-10T20:33:55.189876Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.174542Z","title":"Gpt-4 technical report, 2023","venue":null,"work_id":"b9f90daa-3c2e-4588-b377-2094993be61b","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.160875Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:b4b1509574f05c3d747e58ab65783566a67bb87790766badec161a7a9d302932","observation_id":"feebdaca-2d58-4d68-b35f-94a596c1ce9d","resolution":{"observed_at":"2026-08-10T20:33:55.178240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.161739Z","title":"Gpt-4o system card, 2024","venue":null,"work_id":"aef30f49-f526-4bd4-a9f5-21eb4f8a00c4","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.167146Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:ca6eac4e3df2f76e212497920130ad695adef20c82e7d08401e068da18bfe4ba","observation_id":"68ab66d2-4688-4f8d-8a79-199d155881f5","resolution":{"observed_at":"2026-08-10T20:33:55.166879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.148400Z","title":"Digihuman: A con- versational digital human with facial expressions","venue":null,"work_id":"8ea23ae4-3110-436b-aae0-3dd8ee403771","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.171845Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:4063c46e8926f4d60966924cbdd21a5dfadd4e82fa2c1abc8118856048ca1360","observation_id":"fed2f313-d31c-47f1-a744-9a6b3e0308f9","resolution":{"observed_at":"2026-08-10T20:33:55.152984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.136614Z","title":"Pantic, M","venue":null,"work_id":"32e709de-8d95-42c6-a626-fc33901e8655","year":2005},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.176751Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:3c99198df2b20d3b9aece7366b7a2b6775e1a5d61a07daed599bbd0156c69e03","observation_id":"1d57b4a1-9157-4f8f-a53c-d32938b2a18e","resolution":{"observed_at":"2026-08-10T20:33:55.140334Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.181114Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.181114Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:d91f954b2d4a7b6f93c693254c739aa103d51f2f74b7cc06c5c5e82d60c1c686","observation_id":"5af1b47d-f09a-4c75-9ece-bce449cfd8f5","resolution":{"observed_at":"2026-08-10T20:33:54.181114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.118423Z","title":"Movie description","venue":null,"work_id":"fc6ebef7-cc24-492c-9230-dc34f5cdd816","year":2017},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.186009Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:390fe723344a340e8b83eb3a8afc80208280ea2781e6dcb563226567515e829e","observation_id":"113794d7-32e1-42f4-adb7-8d05ccabfe0a","resolution":{"observed_at":"2026-08-10T20:33:55.121966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.107629Z","title":"Multi-view dynamic facial action unit detection","venue":null,"work_id":"ebefc33d-af95-4c8a-a543-be51ba7d1b88","year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.190865Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:da93518202918528ae387ad5a27e8a60f519836c86e901bab1179dcd4ebd6882","observation_id":"d423450c-b4bf-45a3-a3a2-63e59e6853fb","resolution":{"observed_at":"2026-08-10T20:33:55.111532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.093500Z","title":"Deep adaptive attention for joint facial action unit detection and face alignment","venue":null,"work_id":"0a377fc1-aeac-4496-9269-dcac6c487b77","year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.195507Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:0dab5298ad5e08bf551f7a5d9b56b90d303d9b5aab47a4c91e0bfaba8eb8e131","observation_id":"adc5adc1-c310-4d96-94c4-3e30f01709c8","resolution":{"observed_at":"2026-08-10T20:33:55.098590Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.080982Z","title":"Driver’s emotion and behavior classification system based on internet of things and deep learning for advanced driver assistance system (adas)","venue":null,"work_id":"34494831-7f9c-44c3-9102-db82ed84f75b","year":2022},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.200363Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:6164d284e49f211fb450d2fe10ec29612c37eaadec7a5ce631bdd5d8d448da7d","observation_id":"d57afeef-1816-4ffa-ab42-5dd4bc8d113b","resolution":{"observed_at":"2026-08-10T20:33:55.084813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.070258Z","title":"Gemini: A family of highly capable multi- modal models, 2024","venue":null,"work_id":"74eee480-5326-4af6-9aac-2de481c40bca","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.204861Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e66f694ef4eda988901c3cd9abfdf6f79f5ae00c7b17678b36456d4a8bddda95","observation_id":"e9a31929-d21b-4728-9114-4d844a9aa896","resolution":{"observed_at":"2026-08-10T20:33:55.073989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.060456Z","title":"Qwen2.5: A party of foundation models, 2024","venue":null,"work_id":"360e6452-83ca-4535-8ea1-d27e14629222","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.209214Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:8b4e6450091edfc0c6e6bdf6ed476fdc7eda80227f5461a89d192939e5f39b4d","observation_id":"2954039a-fbb4-45d5-a87f-b04dd8aaa782","resolution":{"observed_at":"2026-08-10T20:33:55.063870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.050036Z","title":"Induced disgust, hap- piness and surprise: an addition to the mmi facial expres- sion database","venue":null,"work_id":"52b22bbe-f8a2-4466-9995-7155c75eb003","year":2010},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.213735Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c8e746afbf62706724a0dcef67df589e4a40dd0a4c91b1fdd9e90bf3a900963f","observation_id":"e2ee8a76-9ba8-42df-b627-373ccf0dfbdd","resolution":{"observed_at":"2026-08-10T20:33:55.053978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.038000Z","title":"Cider: Consensus-based image description evalua- tion","venue":null,"work_id":"848b0105-3e1a-4942-bc00-0b924cde8f2a","year":2015},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.218449Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:fb24cf0fe1f1e8399bad74bf4feb5abadce89beabeb50b50e7335331dee84418","observation_id":"96b16d77-d1c1-42b2-9937-6c6580196ac4","resolution":{"observed_at":"2026-08-10T20:33:55.042966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.024743Z","title":"A survey on the pipeline evolution of facial capture and tracking for digital humans","venue":null,"work_id":"087ee612-73cb-47e0-abe6-7cbb26338b28","year":1917},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.222398Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:03782428fc6a55de1f01863210f982c7b4e9811b4af222e071cef02818ae210d","observation_id":"fe9dd5c5-1566-4bc3-808d-f1c7e92a8f29","resolution":{"observed_at":"2026-08-10T20:33:55.028702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.013998Z","title":"Gross, Kristina H ¨o¨ok, Regan Mandryk, and Petr Slovak","venue":null,"work_id":"dac742cd-dc5b-475c-9521-98accb48f591","year":2022},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.226871Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:7df01646c82f626619d132913ac7daa71de3edf1fd3c13753b02d4c5a2c9644f","observation_id":"f278228a-cd22-4c69-906e-39655cb90a70","resolution":{"observed_at":"2026-08-10T20:33:55.017717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.003526Z","title":"Tarsier: Recipes for training and evaluating large video description models, 2024","venue":null,"work_id":"52d26679-9b51-4354-acec-d2558477530d","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.231149Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:828a8573358a445411f1faf6e02b42851ba7d4d991b1a2d1cd20a715db3e3720","observation_id":"0b0fa2dc-26a1-483e-b33e-485eb0196ed9","resolution":{"observed_at":"2026-08-10T20:33:55.007121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-10T20:33:54.235456Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.235456Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:4c67940532463d05754e737316a051398a5c703d84208052680866c88b4c4844","observation_id":"329bd536-d2db-4793-a525-e6978bf8566a","resolution":{"observed_at":"2026-08-10T20:33:54.235456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.992904Z","title":"Vatex: A large-scale, high- quality multilingual dataset for video-and-language research","venue":null,"work_id":"39e8ac35-5ad1-49f8-85af-507dadcf31a1","year":2019},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.239911Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:da077ec16cd99d581cb3ac3d8ad7b3a1dc73b20b6815298a94cd88a5a844af88","observation_id":"f7cb10f3-2000-433b-95c5-bdfcf0a0d9a0","resolution":{"observed_at":"2026-08-10T20:33:54.996676Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.858284Z","title":"Ferv39k: A large-scale multi-scene dataset for fa- cial expression recognition in videos, 2022","venue":null,"work_id":"1e70381d-a8ff-4241-980c-1044912c2d1e","year":2022},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.244178Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:a3185c1532412b413083c07db75b774f2235aabc38cbee1a98b84a1427ea87e4","observation_id":"8365d89e-3c8f-480b-913e-31a5ee19eee5","resolution":{"observed_at":"2026-08-10T20:33:54.985798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-10T11:36:03.411225Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-10T20:33:54.248879Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.248879Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:050efa2fdeaaee903a113d007090ab05cface9cc8864fa647f6e9d8b98f3f384","observation_id":"17a4cd68-123f-4ab6-a9b6-59c5b94b9864","resolution":{"observed_at":"2026-08-10T20:33:54.248879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.253320Z","title":"Msr-vtt: A large video description dataset for bridging video and language","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.253320Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:9a3acce27e68afc8fa69f4160872ccbfaabf5c3f11654ea28db95dad89cbd4ee","observation_id":"78a57c11-57c5-49b9-a415-014a2697bd64","resolution":{"observed_at":"2026-08-10T20:33:54.253320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.838544Z","title":"Pllava : Parameter-free llava extension from images to videos for video dense captioning, 2024","venue":null,"work_id":"729e22ca-2acc-4d45-b097-fab5ae936613","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.258279Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:21e45e1b85efc7bf522c8b4d28d419e605d54dedbf548da31d2b5e035e1e0e34","observation_id":"b9baf577-4213-4296-898f-e4561352466e","resolution":{"observed_at":"2026-08-10T20:33:54.842118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.262867Z","title":"xgen-mm (blip-3): A family of open large multimodal models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.262867Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:22ce91000c7511c46c09975d8b400049609d2b24f13c9338532d11b6415440e2","observation_id":"b86eb911-55bb-4a55-ad71-570e44686c28","resolution":{"observed_at":"2026-08-10T20:33:54.262867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-10T20:33:54.270705Z","title":"mplug-owl: Modularization empowers large language models with multimodality","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.270705Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:bb3789e08d571ea1dcfe239869ba0b3fcd9d8875648ff70ca2ddf4ee58011faf","observation_id":"a80bd366-e3b8-4c07-bc84-872b9aa404ef","resolution":{"observed_at":"2026-08-10T20:33:54.270705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.811947Z","title":"mplug- owl2: Revolutionizing multi-modal large language model with modality collaboration, 2023","venue":null,"work_id":"a93893c5-c855-47e9-8c2d-f9855faad9be","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.274065Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:9ef1d53d74714afacd7d4bced08525ad82e57eca086b8b0212cbb543516fefe0","observation_id":"7120bbcc-1364-4dbf-9b61-41dc9cd72a46","resolution":{"observed_at":"2026-08-10T20:33:54.814951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.802725Z","title":"Spatio-temporal convolutional features with nested lstm for facial expression recognition.Neurocomputing, 317: 50–57, 2018","venue":null,"work_id":"3f8777c7-379e-4e02-b4b7-bf4db8a853d4","year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.277803Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:da48330ecd57b92f9d4fc2faa504029753204deab30fd4b04b80b77a3c1e767b","observation_id":"99cbcbe2-93a1-4092-83f4-6ed80486ba15","resolution":{"observed_at":"2026-08-10T20:33:54.805828Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.791938Z","title":"Auformer: Vision transformers are parameter-efficient facial action unit detectors","venue":null,"work_id":"2887f477-3f95-4f2f-8fd9-b88d25cb3940","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.281283Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:482b2d1cb738f16e277ce3cb4cc93bc7d98a76fd4395f93fc55a460d3aef5731","observation_id":"1d5ac5bb-0e4d-495e-9633-9bcc71042d74","resolution":{"observed_at":"2026-08-10T20:33:54.796041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-10T20:33:54.285100Z","title":"Video-llama: An instruction-tuned audio-visual language model for video un- derstanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.285100Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:9bfbb7cb637c7dd58f7f9b1786593e74d2d75960decb29f9dd1bc2f4df5da4d0","observation_id":"262d4719-76df-40e9-88b2-d0e486e2a57e","resolution":{"observed_at":"2026-08-10T20:33:54.285100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15105","last_updated":"2023-03-27T11:13:50Z","snapshot_observed_at":"2026-07-06T15:08:22.988867Z","submitted_at":"2023-03-27T11:13:50Z","title":"Vision Transformer with Quadrangle Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.15105","snapshot_observed_at":"2026-08-10T20:33:54.289229Z","title":"Vi- sion transformer with quadrangle attention","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.289229Z"},"links":{"cited_paper":"/paper/2303.15105","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:b737b644cee9d9f88e8e83f40cf8e55796f8039b7fa4e22febe012e4d6433fa5","observation_id":"1ffd1b68-9ee3-4d03-8634-e20268023f95","resolution":{"observed_at":"2026-08-10T20:33:54.289229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.782213Z","title":"Cohn, Shaun Canavan, Michael Reale, Andy Horowitz, Peng Liu, and Jeffrey M","venue":null,"work_id":"18cbc492-a5ef-486b-80b2-3e99ad98c629","year":2014},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.294736Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:97ff2ef08991df4c83fec0d2f8e6601092fcd13b4dbdf0aff4e12c3be63e2799","observation_id":"9812e6b7-9531-4f92-aa79-f3b95bdc06a8","resolution":{"observed_at":"2026-08-10T20:33:54.785326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.771101Z","title":"Llava- next: A strong zero-shot video understanding model, 2024","venue":null,"work_id":"535a8468-c43a-47be-b100-cf75a24d8a03","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.298057Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:074b02b8dd069a7e74248c46288731960a9374cc42d3baa8e1d0853bec014548","observation_id":"ca82cb7d-5634-4986-921c-5a23b8072967","resolution":{"observed_at":"2026-08-10T20:33:54.774625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.759763Z","title":"Facial expression recognition from near- infrared videos","venue":null,"work_id":"279adfd5-ace0-4bf2-ae94-db72e9cdc4c0","year":2011},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.301613Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:4a324544831a15c6728794668aad579b6ec0b8e92f3f240a3219d4ab5f83fd9c","observation_id":"e265d731-5a5a-48a6-beac-ca464787be71","resolution":{"observed_at":"2026-08-10T20:33:54.764001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05067","last_updated":"2026-04-20T07:42:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-09T08:43:57Z","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.05067","snapshot_observed_at":"2026-08-10T20:33:54.304945Z","title":"Llava-octopus: Unlocking instruction-driven adaptive projector fusion for video understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.304945Z"},"links":{"cited_paper":"/paper/2501.05067","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:305f49ce8ea4b61b5cdf7da256adc3a176c1102cc2c8847ac622d6121170946f","observation_id":"eace1683-3784-4d4d-a46a-932b0a4d921c","resolution":{"observed_at":"2026-08-10T20:33:54.304945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.747361Z","title":"Deep region and multi-label learning for facial action unit detec- tion","venue":null,"work_id":"d4a6c64a-0d37-4491-856a-9344ae3f71c7","year":2016},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.308759Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:efb09f0e762d7cb61e6a0ee0c0d3f3a98958d0af1f9b0c3ca6aebf1824a92e56","observation_id":"14e9c2b1-9476-4ab0-b54c-3afe638013e0","resolution":{"observed_at":"2026-08-10T20:33:54.751577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-10T20:33:54.312278Z","title":"Mlvu: A comprehensive benchmark for multi-task long video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.312278Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:6f3b5d93f334e417bacdc861092105f6b30395600dc8f686225ffc2991f69e0f","observation_id":"6d9ec7d2-adf8-4084-9f4a-740fce5b0c11","resolution":{"observed_at":"2026-08-10T20:33:54.312278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.732859Z","title":"Towards automatic learning of procedures from web instructional videos","venue":null,"work_id":"1b03852d-3ad5-4ade-9fd6-786c22fa8c04","year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.316060Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:f0a43e1518f46d16933a86a09acce097610667c7d284446c4feed66886c3abdc","observation_id":"2276db77-201b-497d-a22f-600d3dd25b63","resolution":{"observed_at":"2026-08-10T20:33:54.739302Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-10T20:33:54.320561Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.320561Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:105421fde55362f59bcc0bd44159199575dc630ec796a0082bb8a003725b50c7","observation_id":"30e8f1e5-610c-4d29-a881-238bc6134e5d","resolution":{"observed_at":"2026-08-10T20:33:54.320561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-11T15:14:22.113908Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness"},"reference_resolution":{"displayed":95,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":52,"verified_exact":0,"verified_fuzzy":43},"total_outbound_references":95},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 95 of 95 outbound references and 4 inbound Pith citation observations for arXiv:2501.07978."}