{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ERJ4Z4PX75CTGHFQQCWB3XDMWN","short_pith_number":"pith:ERJ4Z4PX","schema_version":"1.0","canonical_sha256":"2453ccf1f7ff45331cb080ac1ddc6cb34a344b64a5503de6791ebd25d58de44c","source":{"kind":"arxiv","id":"2411.15628","version":1},"attestation_state":"computed","paper":{"title":"ACE: Action Concept Enhancement of Video-Language Models in Procedural Videos","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Behzad Darisuh, Isht Dwivedi, Nakul Agarwal, Reza Ghoddoosian","submitted_at":"2024-11-23T18:49:49Z","abstract_excerpt":"Vision-language models (VLMs) are capable of recognizing unseen actions. However, existing VLMs lack intrinsic understanding of procedural action concepts. Hence, they overfit to fixed labels and are not invariant to unseen action synonyms. To address this, we propose a simple fine-tuning technique, Action Concept Enhancement (ACE), to improve the robustness and concept understanding of VLMs in procedural action classification. ACE continually incorporates augmented action synonyms and negatives in an auxiliary classification loss by stochastically replacing fixed labels during training. This "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15628","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-23T18:49:49Z","cross_cats_sorted":[],"title_canon_sha256":"6daf644832da20e70230a6aae55633d092fe41969b4f78220a6da198c2633437","abstract_canon_sha256":"202df75fb04ff397eebeadba8e4a6685e7fd4e77ed4592aec5ff162be4363b3c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:38.296101Z","signature_b64":"D7j5+PSs6t/4c2QGA+AnYE9HmfOgx4IUiW3MOIqT6P7b8FnrSFfk3HmKmE6d2LKe9x9gZ6xgrusxGGD84pvgDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2453ccf1f7ff45331cb080ac1ddc6cb34a344b64a5503de6791ebd25d58de44c","last_reissued_at":"2026-07-05T09:39:38.295645Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:38.295645Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ACE: Action Concept Enhancement of Video-Language Models in Procedural Videos","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Behzad Darisuh, Isht Dwivedi, Nakul Agarwal, Reza Ghoddoosian","submitted_at":"2024-11-23T18:49:49Z","abstract_excerpt":"Vision-language models (VLMs) are capable of recognizing unseen actions. However, existing VLMs lack intrinsic understanding of procedural action concepts. Hence, they overfit to fixed labels and are not invariant to unseen action synonyms. To address this, we propose a simple fine-tuning technique, Action Concept Enhancement (ACE), to improve the robustness and concept understanding of VLMs in procedural action classification. ACE continually incorporates augmented action synonyms and negatives in an auxiliary classification loss by stochastically replacing fixed labels during training. This "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15628","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15628/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15628","created_at":"2026-07-05T09:39:38.295704+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15628v1","created_at":"2026-07-05T09:39:38.295704+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15628","created_at":"2026-07-05T09:39:38.295704+00:00"},{"alias_kind":"pith_short_12","alias_value":"ERJ4Z4PX75CT","created_at":"2026-07-05T09:39:38.295704+00:00"},{"alias_kind":"pith_short_16","alias_value":"ERJ4Z4PX75CTGHFQ","created_at":"2026-07-05T09:39:38.295704+00:00"},{"alias_kind":"pith_short_8","alias_value":"ERJ4Z4PX","created_at":"2026-07-05T09:39:38.295704+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ERJ4Z4PX75CTGHFQQCWB3XDMWN","json":"https://pith.science/pith/ERJ4Z4PX75CTGHFQQCWB3XDMWN.json","graph_json":"https://pith.science/api/pith-number/ERJ4Z4PX75CTGHFQQCWB3XDMWN/graph.json","events_json":"https://pith.science/api/pith-number/ERJ4Z4PX75CTGHFQQCWB3XDMWN/events.json","paper":"https://pith.science/paper/ERJ4Z4PX"},"agent_actions":{"view_html":"https://pith.science/pith/ERJ4Z4PX75CTGHFQQCWB3XDMWN","download_json":"https://pith.science/pith/ERJ4Z4PX75CTGHFQQCWB3XDMWN.json","view_paper":"https://pith.science/paper/ERJ4Z4PX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15628&json=true","fetch_graph":"https://pith.science/api/pith-number/ERJ4Z4PX75CTGHFQQCWB3XDMWN/graph.json","fetch_events":"https://pith.science/api/pith-number/ERJ4Z4PX75CTGHFQQCWB3XDMWN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ERJ4Z4PX75CTGHFQQCWB3XDMWN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ERJ4Z4PX75CTGHFQQCWB3XDMWN/action/storage_attestation","attest_author":"https://pith.science/pith/ERJ4Z4PX75CTGHFQQCWB3XDMWN/action/author_attestation","sign_citation":"https://pith.science/pith/ERJ4Z4PX75CTGHFQQCWB3XDMWN/action/citation_signature","submit_replication":"https://pith.science/pith/ERJ4Z4PX75CTGHFQQCWB3XDMWN/action/replication_record"}},"created_at":"2026-07-05T09:39:38.295704+00:00","updated_at":"2026-07-05T09:39:38.295704+00:00"}