{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:76U4PEG44KAXY4L4I7XH2MCGIL","short_pith_number":"pith:76U4PEG4","schema_version":"1.0","canonical_sha256":"ffa9c790dce2817c717c47ee7d304642ea35d9b334bf25159efb838e929a54a8","source":{"kind":"arxiv","id":"2409.00304","version":2},"attestation_state":"computed","paper":{"title":"StimuVAR: Spatiotemporal Stimuli-aware Video Affective Reasoning with Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Faizan Siddiqui, Rama Chellappa, Shao-Yuan Lo, Yang Zhao, Yuxiang Guo","submitted_at":"2024-08-31T00:00:50Z","abstract_excerpt":"Predicting and reasoning how a video would make a human feel is crucial for developing socially intelligent systems. Although Multimodal Large Language Models (MLLMs) have shown impressive video understanding capabilities, they tend to focus more on the semantic content of videos, often overlooking emotional stimuli. Hence, most existing MLLMs fall short in estimating viewers' emotional reactions and providing plausible explanations. To address this issue, we propose StimuVAR, a spatiotemporal Stimuli-aware framework for Video Affective Reasoning (VAR) with MLLMs. StimuVAR incorporates a two-l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.00304","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-31T00:00:50Z","cross_cats_sorted":[],"title_canon_sha256":"045d1e5450bbf688d5b3d8536c9a43745ac6c38ccfe90e5cd72bd4c388906790","abstract_canon_sha256":"d43908b0c2186e2abbee28634ffb728698d0cf28bab2a78995e977f6c4b36943"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:24.458394Z","signature_b64":"O5x2LlKQZBGw0YX32GaA0uIT9sH/1dyjFS2aAwcQ1ib9U8kr5F41cISMb+k6c5TkEUyZXlErk0NTnmoS32RlAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ffa9c790dce2817c717c47ee7d304642ea35d9b334bf25159efb838e929a54a8","last_reissued_at":"2026-07-05T11:14:24.457927Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:24.457927Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StimuVAR: Spatiotemporal Stimuli-aware Video Affective Reasoning with Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Faizan Siddiqui, Rama Chellappa, Shao-Yuan Lo, Yang Zhao, Yuxiang Guo","submitted_at":"2024-08-31T00:00:50Z","abstract_excerpt":"Predicting and reasoning how a video would make a human feel is crucial for developing socially intelligent systems. Although Multimodal Large Language Models (MLLMs) have shown impressive video understanding capabilities, they tend to focus more on the semantic content of videos, often overlooking emotional stimuli. Hence, most existing MLLMs fall short in estimating viewers' emotional reactions and providing plausible explanations. To address this issue, we propose StimuVAR, a spatiotemporal Stimuli-aware framework for Video Affective Reasoning (VAR) with MLLMs. StimuVAR incorporates a two-l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.00304","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.00304/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.00304","created_at":"2026-07-05T11:14:24.457979+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.00304v2","created_at":"2026-07-05T11:14:24.457979+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.00304","created_at":"2026-07-05T11:14:24.457979+00:00"},{"alias_kind":"pith_short_12","alias_value":"76U4PEG44KAX","created_at":"2026-07-05T11:14:24.457979+00:00"},{"alias_kind":"pith_short_16","alias_value":"76U4PEG44KAXY4L4","created_at":"2026-07-05T11:14:24.457979+00:00"},{"alias_kind":"pith_short_8","alias_value":"76U4PEG4","created_at":"2026-07-05T11:14:24.457979+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25325","citing_title":"Omni-Perception Policy Optimization for Multimodal Emotion Reasoning","ref_index":128,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/76U4PEG44KAXY4L4I7XH2MCGIL","json":"https://pith.science/pith/76U4PEG44KAXY4L4I7XH2MCGIL.json","graph_json":"https://pith.science/api/pith-number/76U4PEG44KAXY4L4I7XH2MCGIL/graph.json","events_json":"https://pith.science/api/pith-number/76U4PEG44KAXY4L4I7XH2MCGIL/events.json","paper":"https://pith.science/paper/76U4PEG4"},"agent_actions":{"view_html":"https://pith.science/pith/76U4PEG44KAXY4L4I7XH2MCGIL","download_json":"https://pith.science/pith/76U4PEG44KAXY4L4I7XH2MCGIL.json","view_paper":"https://pith.science/paper/76U4PEG4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.00304&json=true","fetch_graph":"https://pith.science/api/pith-number/76U4PEG44KAXY4L4I7XH2MCGIL/graph.json","fetch_events":"https://pith.science/api/pith-number/76U4PEG44KAXY4L4I7XH2MCGIL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/76U4PEG44KAXY4L4I7XH2MCGIL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/76U4PEG44KAXY4L4I7XH2MCGIL/action/storage_attestation","attest_author":"https://pith.science/pith/76U4PEG44KAXY4L4I7XH2MCGIL/action/author_attestation","sign_citation":"https://pith.science/pith/76U4PEG44KAXY4L4I7XH2MCGIL/action/citation_signature","submit_replication":"https://pith.science/pith/76U4PEG44KAXY4L4I7XH2MCGIL/action/replication_record"}},"created_at":"2026-07-05T11:14:24.457979+00:00","updated_at":"2026-07-05T11:14:24.457979+00:00"}