{"as_of":"2026-08-07T05:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bc97585c40d01b711d470177d9a5933badd5543d2dd3e4e19ff9b5084899e93c","coverage":[{"denominator":94,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":94,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T13:07:27.714660Z","state":"measured"},{"denominator":94,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":94,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.21015/citation-record","integrity":"/paper/2507.21015/integrity","json":"/paper/2507.21015/citation-record.json","paper":"/paper/2507.21015"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.592232Z","title":"Society of mind","venue":null,"work_id":null,"year":1986},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.592232Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:3326456bd2bf56472f304d3f00041912a75ac6a44b75391863e4525026851367","observation_id":"ef8048ad-0f81-48e2-95bd-6228b7193da3","resolution":{"observed_at":"2026-08-06T13:07:26.592232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.661748Z","title":"Emotion recognition in human-computer interaction","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.661748Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:5a9c4fb2d3e5c971504f8c4b5dbd070de34f26628bc6ddf17a834b3572e0b4e6","observation_id":"83c82526-bf1a-4ee5-a186-383f493b7c67","resolution":{"observed_at":"2026-08-06T13:07:26.661748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.741061Z","title":"An overview of emotion in artificial intelligence","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.741061Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:5233f4daf0515f2bf30520a9dbbfad609235b70a1822afb25f15147d0ca1914c","observation_id":"f491c3aa-26ad-479e-9029-d58d3660f428","resolution":{"observed_at":"2026-08-06T13:07:26.741061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.793886Z","title":"Deep facial expression recognition: A survey","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.793886Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:5b75b535afe4fa5aa32d5575220d967180970472b2da2cf0db65405ff39e13e8","observation_id":"e807f6d5-3610-46d3-b508-83399eee5a45","resolution":{"observed_at":"2026-08-06T13:07:26.793886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.873360Z","title":"A survey on facial emotion recognition techniques: A state-of-the-art literature review","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.873360Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8808868eda75a226c2c86730e384fe3186a48f4f61707aa4d49edf155e0924be","observation_id":"1dba552a-29c9-4621-8fd8-d36930e6142a","resolution":{"observed_at":"2026-08-06T13:07:26.873360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.940239Z","title":"Understanding deep learning techniques for recognition of human emotions using facial expressions: A comprehensive survey","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.940239Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:b96ec85bfb754d9421f1e20a9ad8b7d44166988918f9d7016faecc0aeccbcbce","observation_id":"a0818c7a-b929-4123-a7ab-abe8debb2fcc","resolution":{"observed_at":"2026-08-06T13:07:26.940239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.056038Z","title":"Facial micro-expressions: An overview","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.056038Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:71792c2a175bfc2ae9f9147122734d9021ea41fad9e231fd0253c68eaaf5e35a","observation_id":"1a786f0b-de19-4653-b951-d668d2a57ee0","resolution":{"observed_at":"2026-08-06T13:07:27.056038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.176560Z","title":"A model of the perception of facial expressions of emotion by humans: Research overview and perspectives","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.176560Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:ef6f1a97b87eef6c2ed8e86089cf2adbdc49537ef92ebce28c962f0db04271d3","observation_id":"2c240e3f-d6e6-4c7d-abab-3ca7006dbceb","resolution":{"observed_at":"2026-08-06T13:07:27.176560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.230927Z","title":"Deep learning for human affect recognition: Insights and new developments","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.230927Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:f326f902f00f54865682ca5c8e98b49252b63306ff292f2f44d00f5360b9fa23","observation_id":"d97ce718-b408-409d-902e-c023dde91c0f","resolution":{"observed_at":"2026-08-06T13:07:27.230927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.295239Z","title":"A review of affective computing: From unimodal analysis to multimodal fusion","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.295239Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:ef2ba37bb78bf3a80dfbff6c7ef28183a13e9bd2c87e529c19bf62bd1b03a72f","observation_id":"dd551dc5-478a-40a1-ac56-ad3a6f194767","resolution":{"observed_at":"2026-08-06T13:07:27.295239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.299794Z","title":"An argument for basic emotions","venue":null,"work_id":null,"year":1992},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.299794Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:e2414881ab4cfe15c1a8739ec8731f21729ed32de8654f9d479caa52ce3277be","observation_id":"a13612c8-151f-425a-bf05-fd6c844a98a8","resolution":{"observed_at":"2026-08-06T13:07:27.299794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.304830Z","title":"A circumplex model of affect","venue":null,"work_id":null,"year":1980},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.304830Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8d48e2922f99017d02c52427207ab723b8656a4db1bd42c8ed02f499d8457647","observation_id":"e43912a0-793e-476a-ab88-55e0bc60b856","resolution":{"observed_at":"2026-08-06T13:07:27.304830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01495","last_updated":"2025-05-07T13:05:04Z","snapshot_observed_at":"2026-07-06T19:25:56.114976Z","submitted_at":"2024-10-02T12:45:09Z","title":"OV-MER: Towards Open-Vocabulary Multimodal Emotion Recognition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01495","snapshot_observed_at":"2026-08-06T13:07:27.309698Z","title":"Open-vocabulary multimodal emotion recognition: Dataset, metric, and benchmark","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.309698Z"},"links":{"cited_paper":"/paper/2410.01495","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:d174786faff4146f90a42873172d92434e0cd64eaf66106e06b3b13eada83565","observation_id":"b75452e2-5d5f-492f-b21f-126db4a70919","resolution":{"observed_at":"2026-08-06T13:07:27.309698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-06T13:07:27.314585Z","title":"Gpt-4o system card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.314585Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:7e58fc008f50192ba1dd6f6a741b0fa280b7582ceba9975034172868a16b211c","observation_id":"8450f0b3-3ee7-4861-a546-d95f9703d13b","resolution":{"observed_at":"2026-08-06T13:07:27.314585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.320211Z","title":"Self-report captures 27 distinct categories of emotion bridged by continuous gradients","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.320211Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:21171f78217685a75f0f19b10264a02df8e837c47a1b2e574968d4300baeba47","observation_id":"3f4eb40c-0771-4fde-b84e-45e9ef730be2","resolution":{"observed_at":"2026-08-06T13:07:27.320211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.324876Z","title":"The language of emotion","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.324876Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:990084016cac387e114e77763fb4a30dcfb36277ced3c0eb238433631fe55ab8","observation_id":"8e58cf6f-20f1-4b0b-bed8-4ba023332c5e","resolution":{"observed_at":"2026-08-06T13:07:27.324876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.186200Z","title":"The role of language in emotion: Predictions from psychological constructionism","venue":null,"work_id":"df1658be-f5fa-4a4b-a173-b6f9c1f20c5f","year":2015},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.329266Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:20ed9e16982bc63d76824ce53b0bf839feeb4d0cd481e63f7f573e099984d97d","observation_id":"474ccc17-737c-4a19-b8e1-b08742ecb011","resolution":{"observed_at":"2026-08-06T13:07:29.190794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.170017Z","title":"Describe your facial expressions by linking image encoders and large language models","venue":null,"work_id":"f40598b9-dc32-4b7d-aaef-916bec08925e","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.333575Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:16928c0efc247f800091bc0014bb315895245f9dbe2bd50b35b2c23d456f3d6f","observation_id":"613824fc-1400-4cf0-ae0d-3f896e0f5bce","resolution":{"observed_at":"2026-08-06T13:07:29.174981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.153389Z","title":"Facial affective behavior analysis with instruction tuning","venue":null,"work_id":"c25d6924-36aa-49cd-b45c-aaa15c5f9611","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.337588Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:cf78b352ac72c628cb2e06940f435bb28e9b695d2020e0c80e7adc6daf8e6129","observation_id":"6747419b-2eb9-4c31-8d2e-fbce8c61250f","resolution":{"observed_at":"2026-08-06T13:07:29.158768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.342458Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.342458Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:4bc1ad0af3c6813007907ce80bcc1504a7d26be43619367f78b896af80ad4e75","observation_id":"00e97ee9-0119-45ec-8c4d-9cfcaf6414f9","resolution":{"observed_at":"2026-08-06T13:07:27.342458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.128004Z","title":"Emoclip: A vision-language method for zero-shot video facial expression recognition","venue":null,"work_id":"b856731f-ae95-48f6-8cce-f7eb58107f49","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.346762Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:2209e972ec3e4863c4cbc808d7dea2605dba08bff1b30f4e41d3a72455a81426","observation_id":"27759557-c862-47df-b6a3-0975c4adcc68","resolution":{"observed_at":"2026-08-06T13:07:29.132614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.113077Z","title":"Flip-80m: 80 million visual-linguistic pairs for facial language-image pre-training","venue":null,"work_id":"7859a851-e8bf-4f21-9daa-cf1679292132","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.350761Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:0202b78d8ecead222f0691c974800f93743921de609a900095f64ac838f033a6","observation_id":"63b2af78-2d5c-48e6-a36e-d6e31ff541ff","resolution":{"observed_at":"2026-08-06T13:07:29.117499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.097446Z","title":"Enhancing zero-shot facial expression recognition by llm knowledge transfer","venue":null,"work_id":"550e0e6a-ebb7-4faf-a9ca-54a329981acd","year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.355611Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:5941520c76c81be4dfa6d0aef5617cf5848dba3ae103ed1e64e8a0f64e0eca63","observation_id":"af6ee1a3-6628-434e-8d7d-7bacdc929a0f","resolution":{"observed_at":"2026-08-06T13:07:29.102546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.360923Z","title":"Facexbench: Evaluating multimodal llms on face understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.360923Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:373368797597cc2c5f5475e7d6f7165d80622780e627bce109990ea5ddeee315","observation_id":"796b78a5-1e88-4291-95cf-94103d0cd1ba","resolution":{"observed_at":"2026-08-06T13:07:27.360923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.365402Z","title":"Face-human-bench: A comprehensive benchmark of face and human understanding for multi-modal assistants","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.365402Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:42863728a9f31c734a6739174a9ff2d178bb88ea049cda2282507a44bbf76019","observation_id":"101f4a02-2524-4b60-84a9-2fee07ea7bc7","resolution":{"observed_at":"2026-08-06T13:07:27.365402Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.370480Z","title":"Gpt-4v with emotion: A zero-shot benchmark for generalized emotion recognition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.370480Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:bc895797cc8e876fa4ea519bee764206b675d1470886bb12a3093f5883fe59eb","observation_id":"9093d8a4-4c21-4974-bc0d-27f203cb14d5","resolution":{"observed_at":"2026-08-06T13:07:27.370480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-06T13:07:27.375535Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.375535Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:4b48923947abf908304642be2ad630c13608f50fb10c6808d9e1a37ddd1ce522","observation_id":"3b546bfb-8e31-4df9-9027-375b17c4b38a","resolution":{"observed_at":"2026-08-06T13:07:27.375535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.071546Z","title":"Occlusion aware facial expression recognition using cnn with attention mechanism","venue":null,"work_id":"d7a8f3da-8418-46df-bba8-b7701dfb94f6","year":2018},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.385225Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:a5d8a66bfd5c1c2e3274ceac6a750d375967ccac739c72a65057f378cffdc5c9","observation_id":"0d4882c9-0f73-42ab-ba71-9383e160156d","resolution":{"observed_at":"2026-08-06T13:07:29.075874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.055875Z","title":"Region attention networks for pose and occlusion robust facial expression recognition","venue":null,"work_id":"35c1b493-580b-42db-83a2-dbb598dda76a","year":2020},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.390540Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:3b67fda5ac0a196ffdd47c675e97218ee388ea192a31b9d322e3bb00ea80084c","observation_id":"c26d7ed2-1a9b-4b0f-beab-c05733330d07","resolution":{"observed_at":"2026-08-06T13:07:29.061244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.040029Z","title":"Learning deep global multi-scale and local attention features for facial expression recognition in the wild","venue":null,"work_id":"3ba27e78-6747-44a2-a506-2fdd4265db2f","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.395215Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:a2ca0d7d853dcaf86893a52868dd529e3834f542803c4799a603bede8dc8a6a5","observation_id":"430b144b-c4bd-4e05-800f-4dd2e61dbb53","resolution":{"observed_at":"2026-08-06T13:07:29.044843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.024850Z","title":"Robust lightweight facial expression recognition network with label distribution training","venue":null,"work_id":"8967b7aa-f1de-4414-b473-ab1eb4490c06","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.399715Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:623876fdb88a13ced32d91df781ebc48881017f1908df65b99d0172b2e1786b8","observation_id":"7e04cd3e-a95e-4130-a8b5-01fc21efedf3","resolution":{"observed_at":"2026-08-06T13:07:29.029929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.009538Z","title":"Facial expression recognition with visual transformers and attentional selective fusion","venue":null,"work_id":"97ec0a49-af22-451a-86df-b021f6011508","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.407722Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:5b6715e7bb5a312cf4de1302e3001348318c79c3c2f727031b113b257f898cf4","observation_id":"80dd1b37-590b-4991-b264-f22b09697a1a","resolution":{"observed_at":"2026-08-06T13:07:29.014568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.993449Z","title":"Transfer: Learning relation-aware facial expression representations with transformers","venue":null,"work_id":"eeff6080-adbb-49fb-8175-f65da44163cf","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.411851Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:77c26c698fb7dc41b5027689f277adee9fde6db11336b83016e2d3958ff59472","observation_id":"c1ea6df3-43d5-425c-a3df-94839e7e39de","resolution":{"observed_at":"2026-08-06T13:07:28.998550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.974742Z","title":"Poster: A pyramid cross-fusion transformer network for facial expression recognition","venue":null,"work_id":"3723528d-839f-4b15-a3f3-8ef5dad37784","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.415946Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:79eae58469a1dee171cbb3e26a82bfcba46a715ecb124d7fba4e1d55f3db39b0","observation_id":"4dbb4180-0854-43b3-a52d-afac2bfb917e","resolution":{"observed_at":"2026-08-06T13:07:28.979206Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.957667Z","title":"Svfap: Self-supervised video facial affect perceiver","venue":null,"work_id":"0175a2c1-37d6-42ff-baab-2da226d4da33","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.420289Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:be6ad105dcc0b2b0fcec323e126fc6cf3fcae41be33948969d12d4f2da822608","observation_id":"334c6763-05d5-4a4d-b5c1-ded126b9604a","resolution":{"observed_at":"2026-08-06T13:07:28.963118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.938456Z","title":"Poster++: A simpler and stronger facial expression recognition network","venue":null,"work_id":"e8dd462d-78f7-4075-b6fa-efbc981f1dcf","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.424789Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:f05ea2c8648b42c4d4c174303378e4b85e031ed23c8e88b4ba729404f0c97d4d","observation_id":"b831ece9-cd9f-4e18-9f18-1e7875b8a1ed","resolution":{"observed_at":"2026-08-06T13:07:28.943246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.921667Z","title":"Reliable crowdsourcing and deep locality-preserving learning for expression recognition in the wild","venue":null,"work_id":"02c0d7bb-f36a-4028-b954-018a9ac574bf","year":2017},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.431081Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:74090996c85a1c382a52ee22ddce436815d3a24d51005c9d113cb29c78a59550","observation_id":"5d933e14-0c5a-4771-8e24-f997f8411e73","resolution":{"observed_at":"2026-08-06T13:07:28.927537Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.904804Z","title":"Suppressing uncertainties for large-scale facial expression recognition","venue":null,"work_id":"1bba0894-94ee-4781-b5ea-a57cec950a7e","year":2020},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.435495Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:1ac6b742cf89dc01b595dd38d111a61507c2ef148e2a30e411d59e295053597d","observation_id":"d4b4681a-434e-4c5f-89b7-9ff5626202a1","resolution":{"observed_at":"2026-08-06T13:07:28.910444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.886278Z","title":"Relative uncertainty learning for facial expression recognition","venue":null,"work_id":"46116927-d49d-4606-b174-4ed549c024d6","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.439812Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:aeadeee61f7593914a908300f0fdaa195e71fc3d3e4f4cb8d16b18a13d8d5f7d","observation_id":"d9df4532-1db3-4df1-9c92-b1a861982c28","resolution":{"observed_at":"2026-08-06T13:07:28.892265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.868839Z","title":"Learning emotion representations from verbal and nonverbal communication","venue":null,"work_id":"746a5111-8979-4a26-99f6-0537c1e65fbb","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.444071Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:9dc650ef2eac57ab563d0dc8bc49db979349e82eb4f8d123917e17b3ef11b12c","observation_id":"ec11e3d6-aea9-44c9-9e48-c902aa3a5b2a","resolution":{"observed_at":"2026-08-06T13:07:28.874331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.853063Z","title":"Collecting large, richly annotated facial-expression databases from movies","venue":null,"work_id":"13041206-54c2-4249-ab55-f4cc0a16c834","year":2012},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.448368Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:f978c255d4d7f7c5c767e8a09d02d4e606a5d1bcac5f0377b0df237feacd558d","observation_id":"668b0632-e627-4b26-ac6b-6ece05f3dba8","resolution":{"observed_at":"2026-08-06T13:07:28.857788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.837472Z","title":"Training deep networks for facial expression recognition with crowd-sourced label distribution","venue":null,"work_id":"7d76a70c-39d2-4a8a-ba1d-1fdde4986c4a","year":2016},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.453976Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:167862bc0a474a4cbc426171864cc9f743a5468e7b85068f939da47eaf61f1fc","observation_id":"53205689-aba6-4690-b331-ea5483dfff10","resolution":{"observed_at":"2026-08-06T13:07:28.842244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.821293Z","title":"Affectnet: A database for facial expression, valence, and arousal computing in the wild","venue":null,"work_id":"3bcc0b4d-2c01-47b1-a8ca-eac3110e0151","year":2017},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.458964Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:e52c1fa3d54e24ef10eb1345aa008c9408395b8b4d1c3a561a6683b0847c3019","observation_id":"d9730ec3-50fe-4659-b3d4-074cc7ab87f4","resolution":{"observed_at":"2026-08-06T13:07:28.825975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.805419Z","title":"Dfew: A large-scale database for recognizing dynamic facial expressions in the wild","venue":null,"work_id":"f008b22d-952f-4f4c-b6b5-e4f590a14e14","year":2020},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.463896Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:585b6093c2a13d61a429cdf92eb21e5c983843b18c1e895d7303b65de30da5f8","observation_id":"2b92e18b-5709-42bd-8711-922081a8989c","resolution":{"observed_at":"2026-08-06T13:07:28.810550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.789228Z","title":"Mafw: A large-scale, multi-modal, compound affective database for dynamic facial expression recognition in the wild","venue":null,"work_id":"d37623f4-33e2-4d6f-bedd-076a2dbc73b6","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.468537Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:30d11cdf9de8fc58d7045a70a6ec8642145440fda4741fc31412beca5958dfc3","observation_id":"e5d8ca84-2fa9-4848-bfc6-bfea2b3c7214","resolution":{"observed_at":"2026-08-06T13:07:28.794142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.771854Z","title":"Ferv39k: A large-scale multi-scene dataset for facial expression recognition in videos","venue":null,"work_id":"bc946e7e-a3de-41c0-b533-136a205408a1","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.472829Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:f176be935348cae9094622a0f9330a502ff1d0f286c7418d904cc4e9f65237bb","observation_id":"859685ba-3115-451b-8b6d-2e6b1be7dad7","resolution":{"observed_at":"2026-08-06T13:07:28.777832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1811.07770","last_updated":"2019-12-13T23:44:20Z","snapshot_observed_at":"2026-08-07T01:58:19.324763Z","submitted_at":"2018-11-11T01:57:15Z","title":"Aff-Wild2: Extending the Aff-Wild Database for Affect Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1811.07770","snapshot_observed_at":"2026-08-06T13:07:27.477365Z","title":"Aff-wild2: Extending the aff-wild database for affect recognition","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.477365Z"},"links":{"cited_paper":"/paper/1811.07770","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:623cd228ce9b3c887a2cc24db73aa65948bde62690342f1b54ff5ae1d59b138f","observation_id":"a6b0f5c3-de36-4929-8dee-e00c386c8737","resolution":{"observed_at":"2026-08-06T13:07:27.477365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.754395Z","title":"Deep affect prediction in-the-wild: Aff-wild database and challenge, deep architectures, and beyond","venue":null,"work_id":"f19540e3-ece7-4eef-afed-b06f66e67083","year":2019},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.482389Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:868bebc1213d8d2849bfec8b9d3fae3581866196f18988ccf1971c3016a844a3","observation_id":"45a3adf5-cf83-46dc-adcf-6d8fe902d8c9","resolution":{"observed_at":"2026-08-06T13:07:28.760052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.737863Z","title":"Compound facial expressions of emotion","venue":null,"work_id":"3f5afb87-f902-4d0a-944a-e9129d52f9c9","year":2014},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.486532Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:ba273dc86bae2f9b186f04152be736f24a854c5d61cad464a2df53a002b5ddb2","observation_id":"68a872fe-c5e7-4beb-b4dc-a0bd6f360814","resolution":{"observed_at":"2026-08-06T13:07:28.742804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.722448Z","title":"Configural information in facial expression perception","venue":null,"work_id":"b584882b-4532-4440-b6be-3f0f9f96fd9d","year":2000},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.491038Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:abcc6cfb49fc814b58ae77ada18ac909e283353c7cdff58940419bd36e59b0b8","observation_id":"bfcf1974-de71-402c-b1eb-b7261af5129e","resolution":{"observed_at":"2026-08-06T13:07:28.727222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.707820Z","title":"Parts and wholes in expression recognition","venue":null,"work_id":"4ea69ae0-1945-4853-8bf1-383a08eb3917","year":2000},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.496626Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:bfe0124ec84bcb094d2573dc6e57f30f34404adb46b900b41b218ae1d3ed77a1","observation_id":"9c4b2c25-bfde-4039-a2a1-ca4eb145d015","resolution":{"observed_at":"2026-08-06T13:07:28.712447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.692053Z","title":"Mixed emotions: Holistic and analytic perception of facial expressions","venue":null,"work_id":"bc7095a6-c5f9-4127-b29a-f6dd2ca3712c","year":2012},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.501743Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:20b0bd273c114b05deafa636619eaa5272e9f6835d6a5553eb03e30089cc75b1","observation_id":"736555e8-e085-4135-9928-40a153ecaa55","resolution":{"observed_at":"2026-08-06T13:07:28.696994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.675969Z","title":"The role of facial movements in emotion recognition","venue":null,"work_id":"88f0b455-e941-443a-8757-238ddced6e95","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.506388Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:d248d897ee98349854134509585436c1f0eceb137cfa6178c8825674031b19e9","observation_id":"c168dbc8-1d94-4a55-baaf-e90bf4d73d38","resolution":{"observed_at":"2026-08-06T13:07:28.680641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.510794Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.510794Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:0b001140138f93e9eb49c52ecf6569f48ce55dead29bb62c780e535a7fa2f651","observation_id":"cc17857f-4354-4c3a-afae-b0d9837b3146","resolution":{"observed_at":"2026-08-06T13:07:27.510794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.515559Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.515559Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:bfff5b16205f209219074d65383b919e8fa93014d3df1735d65d84d67bf1d761","observation_id":"45b1d21e-5166-4104-b63c-945dd671881e","resolution":{"observed_at":"2026-08-06T13:07:27.515559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.519813Z","title":"Reproducible scaling laws for contrastive language-image learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.519813Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:0753d71bfc3b3e6f00a0184ce0715507ed127c937f11829ca4e5b4cd6babdef1","observation_id":"543b6e97-02fa-484a-8e72-b90f188ba575","resolution":{"observed_at":"2026-08-06T13:07:27.519813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.524336Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.524336Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:9ff329b979ad39b41bca03f6d16a7ab39248fc26d561eadf98b59fa831b49d6f","observation_id":"16766737-fb2d-4040-aa01-201cb94a8304","resolution":{"observed_at":"2026-08-06T13:07:27.524336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15389","last_updated":"2023-03-27T17:02:21Z","snapshot_observed_at":"2026-07-06T15:08:34.018146Z","submitted_at":"2023-03-27T17:02:21Z","title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.15389","snapshot_observed_at":"2026-08-06T13:07:27.528659Z","title":"Eva-clip: Improved training techniques for clip at scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.528659Z"},"links":{"cited_paper":"/paper/2303.15389","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:c1492ec0f0becc124dbf1adc5a219bf1ee08076cbc49f243310cf81b48850235","observation_id":"a58c779b-930f-4c26-a463-bf3d2c91a56a","resolution":{"observed_at":"2026-08-06T13:07:27.528659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.616868Z","title":"Demystifying clip data","venue":null,"work_id":"a06b5de1-25ba-46c7-908e-69976bf0bfec","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.534253Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8e21ee8f40fc59dbe1925ad0407edb056006c29a2548354276656a85ddb8e999","observation_id":"e367f1a0-e1a6-41ea-b20d-c15f86c97ebf","resolution":{"observed_at":"2026-08-06T13:07:28.621180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.601141Z","title":"Dreamlip: Language-image pre-training with long captions","venue":null,"work_id":"fb52c6e3-847a-4253-83d4-b0bf37596cce","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.539781Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:742db3e3524dadb8cc708de021d5ac35b001a9cdaaf7cc1f96d56a7e985a0606","observation_id":"296cdcf5-98d0-4dd3-a98d-6a1461eba2ee","resolution":{"observed_at":"2026-08-06T13:07:28.605700Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00740","last_updated":"2025-03-29T12:57:07Z","snapshot_observed_at":"2026-07-06T18:08:21.770341Z","submitted_at":"2024-04-30T01:19:18Z","title":"Modeling Caption Diversity in Contrastive Vision-Language Pretraining","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00740","snapshot_observed_at":"2026-08-06T13:07:27.544588Z","title":"Modeling caption diversity in contrastive vision-language pretraining","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.544588Z"},"links":{"cited_paper":"/paper/2405.00740","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:e22d5e0a96caad1e6dacad26823950f52f05443a2bd0f9125c8e64710fe130b9","observation_id":"172bed96-dc4c-4ace-b307-44b8a6db6d9e","resolution":{"observed_at":"2026-08-06T13:07:27.544588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.585280Z","title":"Improving fine-grained understand- ing in image-text pre-training","venue":null,"work_id":"8a160f06-44b3-41ab-8beb-0660a0c8dcb9","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.549311Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:30e830eb28712b1f59f9504a5e4819aec42ba9c9eb3fc0d5dd496f9076553ce1","observation_id":"96e16bd9-f574-4353-8097-d51cf7a79905","resolution":{"observed_at":"2026-08-06T13:07:28.590425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.554743Z","title":"General facial representation learning in a visual-linguistic manner","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.554743Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:29844882920d6f1e4c3bfc048f0889001a401bec113b06b98951074cf157fb51","observation_id":"2a162f3a-c265-4461-bf56-c969dce59c59","resolution":{"observed_at":"2026-08-06T13:07:27.554743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T13:07:27.559162Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.559162Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:4b0bd4cfc0797227314ebda329421702e789b6294f4b7de6f07eaf87134d12f6","observation_id":"d1da1679-01f1-4495-bf09-5f9664c2e4ca","resolution":{"observed_at":"2026-08-06T13:07:27.559162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.564018Z","title":"Visual instruction tuning.Advances in neural information processing systems, 36:34892–34916, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.564018Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:df425d2931cd19795e8b12498c030dc687c56b8de8f7d3cb0942b4a2507c5f5f","observation_id":"62611581-b192-4cca-832b-9a3429ff96ba","resolution":{"observed_at":"2026-08-06T13:07:27.564018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.548980Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":"13504f6d-b6bd-4179-882e-e9a0215f7470","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.568575Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:be8fcdfa12d24fbcba703baceaf440f7903c74d7d81040e358527690f19b49b2","observation_id":"18510aaf-16b3-425d-8e0a-92146c5d651d","resolution":{"observed_at":"2026-08-06T13:07:28.553765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.573362Z","title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.573362Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:0cf354ff4d03760da539d9c35a900269e762cc2b89319ec041a4e528878af641","observation_id":"a7fe66eb-2834-4485-abe6-a6200aecf3bf","resolution":{"observed_at":"2026-08-06T13:07:27.573362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-06T13:07:27.577897Z","title":"Qwen technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.577897Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:1771d4b08c663c97e5d408ad90dfddefd1b4ce8fb6eb35d30cfab5962a05a23b","observation_id":"0846c629-f9e1-45ee-94ff-fbfa015973c2","resolution":{"observed_at":"2026-08-06T13:07:27.577897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.522421Z","title":"Expllm: Towards chain of thought for facial expression recognition","venue":null,"work_id":"f9803b51-0efd-4cc5-a2f5-fca5b9b22270","year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.583046Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:81777b2e57f726b500b02d65b156c9849e25646d54b92a457dea63adcf4d02f8","observation_id":"b32689e9-eb62-4a25-9e74-442c05fcc889","resolution":{"observed_at":"2026-08-06T13:07:28.527127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.11424","last_updated":"2024-08-21T08:28:40Z","snapshot_observed_at":"2026-08-06T05:36:36.831165Z","submitted_at":"2024-08-21T08:28:40Z","title":"EMO-LLaMA: Enhancing Facial Emotion Understanding with Instruction Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.11424","snapshot_observed_at":"2026-08-06T13:07:27.587777Z","title":"Emo-llama: Enhancing facial emotion understanding with instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.587777Z"},"links":{"cited_paper":"/paper/2408.11424","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8fa1841aa740dbd509158fe71d360b3562c46a3d244b4119b0411cce945ddcf8","observation_id":"91706ae5-009f-4165-9389-758e3f05e0f2","resolution":{"observed_at":"2026-08-06T13:07:27.587777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.505529Z","title":"Emotion-llama: Multimodal emotion recognition and reasoning with instruction tuning","venue":null,"work_id":"8b40948f-c1ee-4880-99fc-d686c3181e32","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.592827Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8bc0ed888ba1510e274f790b7a9cc3d2486d3681f7afab6dcdfe34c397f3e6c1","observation_id":"ad3f9e85-5b32-4f95-bb8f-4e6f97606552","resolution":{"observed_at":"2026-08-06T13:07:28.510694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16566","last_updated":"2025-05-07T13:20:08Z","snapshot_observed_at":"2026-07-06T20:27:04.664873Z","submitted_at":"2025-01-27T23:18:39Z","title":"AffectGPT: A New Dataset, Model, and Benchmark for Emotion Understanding with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16566","snapshot_observed_at":"2026-08-06T13:07:27.597574Z","title":"Affectgpt: A new dataset, model, and benchmark for emotion understanding with multimodal large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.597574Z"},"links":{"cited_paper":"/paper/2501.16566","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:11faa3e6de1852116e8d4358688cb41731c653f95186dddd1702afce1bbe677a","observation_id":"9ff15cd6-87af-4345-9f2f-47c7635f8d50","resolution":{"observed_at":"2026-08-06T13:07:27.597574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.05379","last_updated":"2025-03-10T07:11:14Z","snapshot_observed_at":"2026-07-06T20:48:31.837669Z","submitted_at":"2025-03-07T12:46:42Z","title":"R1-Omni: Explainable Omni-Multimodal Emotion Recognition with Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.05379","snapshot_observed_at":"2026-08-06T13:07:27.602296Z","title":"R1-omni: Explainable omni-multimodal emotion recognition with reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.602296Z"},"links":{"cited_paper":"/paper/2503.05379","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:677b1448f83a51d4bac33b692536956350585faaf9982c8ece36e7c00f1ee27d","observation_id":"dcd2a90e-57dd-493c-898a-6e66e0d1b598","resolution":{"observed_at":"2026-08-06T13:07:27.602296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.487200Z","title":"Generative adversarial network for text-to-face synthesis and manipulation with pretrained bert model","venue":null,"work_id":"988fa545-ac9e-4216-a648-38e0f1c12ab6","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.607503Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:92b84ff5bea725a21fded2d3a50abab50eee68019dcc2303cdf99cfe305da0e2","observation_id":"372f1b4f-0d2c-4d48-9411-8e24d1f8fb4e","resolution":{"observed_at":"2026-08-06T13:07:28.493528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.611815Z","title":"Tedigan: Text-guided diverse face image generation and manipulation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.611815Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:65aeb0011bdebd861c71169e822d6c706cb307b47f42caa39c6dda714cdeca83","observation_id":"d620aab3-8df8-4240-bf39-ae514f932d14","resolution":{"observed_at":"2026-08-06T13:07:27.611815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.458103Z","title":"Talk-to-edit: Fine-grained facial editing via dialog","venue":null,"work_id":"19d9ab30-aa07-4e1e-9a3d-4d6c1f58249d","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.616335Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:cf2ccb9659c8e791de5612c8a46f1e31bc3b840765da2b6c6689cceb3f53b827","observation_id":"bd60d693-e329-40fb-874e-ffc93d0969ba","resolution":{"observed_at":"2026-08-06T13:07:28.463190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08515","last_updated":"2024-07-12T01:19:33Z","snapshot_observed_at":"2026-08-02T05:39:29.403958Z","submitted_at":"2024-07-11T14:00:14Z","title":"15M Multimodal Facial Image-Text Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.08515","snapshot_observed_at":"2026-08-06T13:07:27.621712Z","title":"15m multimodal facial image-text dataset","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.621712Z"},"links":{"cited_paper":"/paper/2407.08515","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:cffb16a61ffa8890d498fd5aec4a0d940ed121ec78fa73c4b6f8a4cadcc7667d","observation_id":"c1535cd7-1311-4ab3-a765-abb76371da81","resolution":{"observed_at":"2026-08-06T13:07:27.621712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08824","last_updated":"2025-03-25T10:14:57Z","snapshot_observed_at":"2026-07-06T17:44:09.081164Z","submitted_at":"2024-03-09T11:16:09Z","title":"Computational Analysis of Stress, Depression and Engagement in Mental Health: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08824","snapshot_observed_at":"2026-08-06T13:07:27.626612Z","title":"Measuring non-typical emotions for mental health: A survey of computational approaches","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.626612Z"},"links":{"cited_paper":"/paper/2403.08824","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:41ca3b2d2eda7fe527ede83e212c6e5659d39c42f836752afbd350269e4afb3d","observation_id":"100cbd00-c651-4570-a040-0f104650aa21","resolution":{"observed_at":"2026-08-06T13:07:27.626612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.443019Z","title":"Recognizing emotion from facial expressions: psychological and neurological mechanisms","venue":null,"work_id":"53e2b549-ac6a-4839-ad5c-b9e3c24c8610","year":2002},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.632099Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8afa4dca301f5ec54255fbaed62dcd377750d25fdff39aba474353c2b2a1ad05","observation_id":"b5516e37-14cc-4f24-95d8-7281e94524eb","resolution":{"observed_at":"2026-08-06T13:07:28.447457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-06T13:07:27.636802Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.636802Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:27e1f3a58185332068f17c194ce04ae15e30c40aa3f22567f06eb953e5a71575","observation_id":"c4942f20-be8d-43ad-9381-17fd57fdd3f2","resolution":{"observed_at":"2026-08-06T13:07:27.636802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.641491Z","title":"Attention is all you need","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.641491Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:de97c2cd3dc063d241eebba65282ca6bc1d67dd271c4dcd4b8f500aecaa9a1fe","observation_id":"49e8119b-58e9-439e-9a72-0efcf4c17de3","resolution":{"observed_at":"2026-08-06T13:07:27.641491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.415813Z","title":"Facial action coding system","venue":null,"work_id":"0f4f2493-e5b4-4f79-96b6-5dcf9c3324d5","year":1978},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.646353Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:dbb7cf297c8c3cc4c406ebc472a6705508692c77ab297ba4a5b6f5676444c025","observation_id":"316b960e-eb7b-47f3-b8bd-6f6fc7c5b4d3","resolution":{"observed_at":"2026-08-06T13:07:28.420990Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.399917Z","title":"Softclip: Softer cross-modal alignment makes clip stronger","venue":null,"work_id":"3e5fd32b-3256-4e59-a4dc-cc1121b0d4a7","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.651724Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:ddd36946de5f8bef9ebc6bee65f12c799a460458747fbd93d01c191c09e089e2","observation_id":"00c31e87-2361-43ec-9cb8-440841c6c4aa","resolution":{"observed_at":"2026-08-06T13:07:28.404940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.382871Z","title":"Cwcl: Cross-modal transfer with continuously weighted contrastive loss","venue":null,"work_id":"c4e294e2-01af-470c-9c9a-da81d393ef65","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.658642Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:ffb196046a144b2867328ba47bcbee8ab062eeeac9c7fa051f550bcac8bd3e5c","observation_id":"c92ff191-88ce-47ea-a195-157a86b31f93","resolution":{"observed_at":"2026-08-06T13:07:28.388027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.365515Z","title":"Generalizable facial expression recognition","venue":null,"work_id":"ac5e2668-2326-4c20-bbcf-32ab7cb95bc3","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.664046Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:f10ebef78205b64ef1f1832c5e82aafddf3515916fb068347ece0051db15a209","observation_id":"c7333be1-82c7-4c11-ab1e-5c7e57932a2d","resolution":{"observed_at":"2026-08-06T13:07:28.371188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.346402Z","title":"Flava: A foundational language and vision alignment model","venue":null,"work_id":"3841f7cf-b2c1-4d6d-8f42-31e66e9b1dd0","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.669757Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:0b7170631abb4d5447880bbc7379993741db431061ea615507936b53a4d6f2bb","observation_id":"c6b01c4a-bc33-4570-bd7d-25213998ac48","resolution":{"observed_at":"2026-08-06T13:07:28.351256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.20717","last_updated":"2024-10-28T04:19:32Z","snapshot_observed_at":"2026-07-06T19:40:34.668379Z","submitted_at":"2024-10-28T04:19:32Z","title":"Face-MLLM: A Large Face Perception Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.20717","snapshot_observed_at":"2026-08-06T13:07:27.675297Z","title":"Face-mllm: A large face perception model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.675297Z"},"links":{"cited_paper":"/paper/2410.20717","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8ca27219face9f51a4ced04a38a95345785a8adbd76aa2d324dc2bae0f0602e5","observation_id":"37e65246-652e-4bcb-92b5-e08e324a1669","resolution":{"observed_at":"2026-08-06T13:07:27.675297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14786","last_updated":"2025-02-20T18:08:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-20T18:08:29Z","title":"SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14786","snapshot_observed_at":"2026-08-06T13:07:27.680305Z","title":"Siglip 2: Multilingual vision-language encoders with improved semantic understanding, localization, and dense features","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.680305Z"},"links":{"cited_paper":"/paper/2502.14786","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:2bb663f245d07166cc854d3ea45345e9b3127e4aaa0be6e51a55bc884c89b6e2","observation_id":"38ee7e1c-45b8-44fe-a6f2-2dbb1596612e","resolution":{"observed_at":"2026-08-06T13:07:27.680305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.327547Z","title":"Learn from all: Erasing attention consistency for noisy label facial expression recognition","venue":null,"work_id":"6a282ab9-b61c-4938-89f9-fcb1e0b8d3b7","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.686607Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:3f1f8b915d440acdaa0292331c53f18553ef54550acd5833d5f039b7f29c8f71","observation_id":"4d20077b-717a-4e68-84af-bfacd77fd4f5","resolution":{"observed_at":"2026-08-06T13:07:28.333222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.308993Z","title":"Latent-ofer: Detect, mask, and reconstruct with latent vectors for occluded facial expression recognition","venue":null,"work_id":"63e735e2-78fa-40f6-9e92-7a89261abce4","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.692832Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:4241e6878472eacfc32b103bb77a6e08ea2c5b7b3f477715415c937952b8f5bd","observation_id":"0c7ce87b-05cb-44d0-a49c-09dfc5bd7168","resolution":{"observed_at":"2026-08-06T13:07:28.314024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.288809Z","title":"From static to dynamic: Adapting landmark-aware image models for facial expression recognition in videos","venue":null,"work_id":"9164280e-c3d2-46b4-8984-b6412ac7e02e","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.698844Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:032530cbfffe713fd72331d65d829bd527f2d7f4a40ba421b2abd996dab5a5e0","observation_id":"0ddd7f09-5bdb-4044-92eb-3384ca30156a","resolution":{"observed_at":"2026-08-06T13:07:28.295881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.272405Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"9ba6373e-18f8-493b-88ba-bb60c51ca05d","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.705145Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:92465edb20f2008c229b16569cea35ab869d34733517b4badd887b7d37b08a52","observation_id":"84411ab7-998a-4dea-b93e-156964029dea","resolution":{"observed_at":"2026-08-06T13:07:28.277285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.256283Z","title":"Expanding language-image pretrained models for general video recognition","venue":null,"work_id":"ca51a836-7950-4017-a47d-313018b9cd45","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.709988Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:522ca61ab1f6bd64f5804a775bf05ee02cdf3e0dd121433da99dea09e8316304","observation_id":"bb9fa76a-8359-49e1-a6b9-59488c37731a","resolution":{"observed_at":"2026-08-06T13:07:28.261598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.236841Z","title":"Learning to prompt for vision-language models","venue":null,"work_id":"1c4e9b3c-37d7-4a3e-a75a-77c9a4b60be4","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.714660Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:a38998d81c2c4e8112460c93da3b883e3b30ac92160403a6b0e1c73200a87164","observation_id":"30c11dac-bc73-4cf0-8cae-94b246a1bbc4","resolution":{"observed_at":"2026-08-06T13:07:28.243764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T13:07:23.473412Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions"},"reference_resolution":{"displayed":94,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":43,"verified_exact":0,"verified_fuzzy":51},"total_outbound_references":94},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 94 of 94 outbound references and 0 inbound Pith citation observations for arXiv:2507.21015."}