{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SRLBLBK6XR2VGI7NHEOEDPGEUK","short_pith_number":"pith:SRLBLBK6","schema_version":"1.0","canonical_sha256":"945615855ebc755323ed391c41bcc4a2b896e68fedebc19c32f3462bd8423bd7","source":{"kind":"arxiv","id":"2309.15785","version":2},"attestation_state":"computed","paper":{"title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chen Li, Ge Li, Ruyang Liu, Thomas H. Li, Ying Shan, Yixiao Ge","submitted_at":"2023-09-27T16:58:35Z","abstract_excerpt":"The recent progress in Large Language Models (LLM) has spurred various advancements in image-language conversation agents, while how to build a proficient video-based dialogue system is still under exploration. Considering the extensive scale of LLM and visual backbone, minimal GPU memory is left for facilitating effective temporal modeling, which is crucial for comprehending and providing feedback on videos. To this end, we propose Branching Temporal Adapter (BT-Adapter), a novel method for extending image-language pretrained models into the video domain. Specifically, BT-Adapter serves as a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.15785","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-09-27T16:58:35Z","cross_cats_sorted":[],"title_canon_sha256":"55fa6a12a082a5b50d3d9a8cbefeede56b3b10578e5cb13b62c7d6f60f6439a4","abstract_canon_sha256":"deb84ec51a5c349add337fe9127ee49b9e6b68d410da50e290fe7c106c3c2200"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:37:14.545068Z","signature_b64":"f6myeeXxHP7FVqe02j2haUgKLItjNTjywChHaDZbGbxHJi2gNrRaUUSr+OuWHUPRJTsY0WGGy7yNEjVU/vgeAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"945615855ebc755323ed391c41bcc4a2b896e68fedebc19c32f3462bd8423bd7","last_reissued_at":"2026-07-05T08:37:14.544552Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:37:14.544552Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chen Li, Ge Li, Ruyang Liu, Thomas H. Li, Ying Shan, Yixiao Ge","submitted_at":"2023-09-27T16:58:35Z","abstract_excerpt":"The recent progress in Large Language Models (LLM) has spurred various advancements in image-language conversation agents, while how to build a proficient video-based dialogue system is still under exploration. Considering the extensive scale of LLM and visual backbone, minimal GPU memory is left for facilitating effective temporal modeling, which is crucial for comprehending and providing feedback on videos. To this end, we propose Branching Temporal Adapter (BT-Adapter), a novel method for extending image-language pretrained models into the video domain. Specifically, BT-Adapter serves as a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.15785","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.15785/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.15785","created_at":"2026-07-05T08:37:14.544614+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.15785v2","created_at":"2026-07-05T08:37:14.544614+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.15785","created_at":"2026-07-05T08:37:14.544614+00:00"},{"alias_kind":"pith_short_12","alias_value":"SRLBLBK6XR2V","created_at":"2026-07-05T08:37:14.544614+00:00"},{"alias_kind":"pith_short_16","alias_value":"SRLBLBK6XR2VGI7N","created_at":"2026-07-05T08:37:14.544614+00:00"},{"alias_kind":"pith_short_8","alias_value":"SRLBLBK6","created_at":"2026-07-05T08:37:14.544614+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2404.16994","citing_title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SRLBLBK6XR2VGI7NHEOEDPGEUK","json":"https://pith.science/pith/SRLBLBK6XR2VGI7NHEOEDPGEUK.json","graph_json":"https://pith.science/api/pith-number/SRLBLBK6XR2VGI7NHEOEDPGEUK/graph.json","events_json":"https://pith.science/api/pith-number/SRLBLBK6XR2VGI7NHEOEDPGEUK/events.json","paper":"https://pith.science/paper/SRLBLBK6"},"agent_actions":{"view_html":"https://pith.science/pith/SRLBLBK6XR2VGI7NHEOEDPGEUK","download_json":"https://pith.science/pith/SRLBLBK6XR2VGI7NHEOEDPGEUK.json","view_paper":"https://pith.science/paper/SRLBLBK6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.15785&json=true","fetch_graph":"https://pith.science/api/pith-number/SRLBLBK6XR2VGI7NHEOEDPGEUK/graph.json","fetch_events":"https://pith.science/api/pith-number/SRLBLBK6XR2VGI7NHEOEDPGEUK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SRLBLBK6XR2VGI7NHEOEDPGEUK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SRLBLBK6XR2VGI7NHEOEDPGEUK/action/storage_attestation","attest_author":"https://pith.science/pith/SRLBLBK6XR2VGI7NHEOEDPGEUK/action/author_attestation","sign_citation":"https://pith.science/pith/SRLBLBK6XR2VGI7NHEOEDPGEUK/action/citation_signature","submit_replication":"https://pith.science/pith/SRLBLBK6XR2VGI7NHEOEDPGEUK/action/replication_record"}},"created_at":"2026-07-05T08:37:14.544614+00:00","updated_at":"2026-07-05T08:37:14.544614+00:00"}