{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BXI4CWW37Q2LBIXEJVYQKSWDUW","short_pith_number":"pith:BXI4CWW3","schema_version":"1.0","canonical_sha256":"0dd1c15adbfc34b0a2e44d71054ac3a583fdcf264f687331c9b827da1073bd3a","source":{"kind":"arxiv","id":"2308.10253","version":2},"attestation_state":"computed","paper":{"title":"StableLLaVA: Enhanced Visual Instruction Tuning with Synthesized Image-Dialogue Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bin Fu, Chi Zhang, Chunhua Shen, Gang Yu, Guosheng Lin, Ling Chen, Yanda Li, Yunchao Wei, Zhibin Wang","submitted_at":"2023-08-20T12:43:52Z","abstract_excerpt":"The remarkable multimodal capabilities demonstrated by OpenAI's GPT-4 have sparked significant interest in the development of multimodal Large Language Models (LLMs). A primary research objective of such models is to align visual and textual modalities effectively while comprehending human instructions. Current methodologies often rely on annotations derived from benchmark datasets to construct image-dialogue datasets for training purposes, akin to instruction tuning in LLMs. However, these datasets often exhibit domain bias, potentially constraining the generative capabilities of the models. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.10253","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-08-20T12:43:52Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"4e58076b6e9a3e5f3d8fdf4b85953781d5cffa8aefb25606e84da90da6c8d579","abstract_canon_sha256":"a46000093ed8611036d79a2d2bd3bc0ff20a158c07a10ccc042e1e10a46b81d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:28:29.419089Z","signature_b64":"B0naXva9tePHXC4lzLQLvMkuoLJcJJ/YhaTTkatgIO6AnLG8+wDSSjaxTtXsmLog0EXRW+3i/7+dEVW5lGHfDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0dd1c15adbfc34b0a2e44d71054ac3a583fdcf264f687331c9b827da1073bd3a","last_reissued_at":"2026-07-05T07:28:29.418559Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:28:29.418559Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StableLLaVA: Enhanced Visual Instruction Tuning with Synthesized Image-Dialogue Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bin Fu, Chi Zhang, Chunhua Shen, Gang Yu, Guosheng Lin, Ling Chen, Yanda Li, Yunchao Wei, Zhibin Wang","submitted_at":"2023-08-20T12:43:52Z","abstract_excerpt":"The remarkable multimodal capabilities demonstrated by OpenAI's GPT-4 have sparked significant interest in the development of multimodal Large Language Models (LLMs). A primary research objective of such models is to align visual and textual modalities effectively while comprehending human instructions. Current methodologies often rely on annotations derived from benchmark datasets to construct image-dialogue datasets for training purposes, akin to instruction tuning in LLMs. However, these datasets often exhibit domain bias, potentially constraining the generative capabilities of the models. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.10253","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.10253/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.10253","created_at":"2026-07-05T07:28:29.418623+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.10253v2","created_at":"2026-07-05T07:28:29.418623+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.10253","created_at":"2026-07-05T07:28:29.418623+00:00"},{"alias_kind":"pith_short_12","alias_value":"BXI4CWW37Q2L","created_at":"2026-07-05T07:28:29.418623+00:00"},{"alias_kind":"pith_short_16","alias_value":"BXI4CWW37Q2LBIXE","created_at":"2026-07-05T07:28:29.418623+00:00"},{"alias_kind":"pith_short_8","alias_value":"BXI4CWW3","created_at":"2026-07-05T07:28:29.418623+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2312.13771","citing_title":"AppAgent: Multimodal Agents as Smartphone Users","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2310.14566","citing_title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BXI4CWW37Q2LBIXEJVYQKSWDUW","json":"https://pith.science/pith/BXI4CWW37Q2LBIXEJVYQKSWDUW.json","graph_json":"https://pith.science/api/pith-number/BXI4CWW37Q2LBIXEJVYQKSWDUW/graph.json","events_json":"https://pith.science/api/pith-number/BXI4CWW37Q2LBIXEJVYQKSWDUW/events.json","paper":"https://pith.science/paper/BXI4CWW3"},"agent_actions":{"view_html":"https://pith.science/pith/BXI4CWW37Q2LBIXEJVYQKSWDUW","download_json":"https://pith.science/pith/BXI4CWW37Q2LBIXEJVYQKSWDUW.json","view_paper":"https://pith.science/paper/BXI4CWW3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.10253&json=true","fetch_graph":"https://pith.science/api/pith-number/BXI4CWW37Q2LBIXEJVYQKSWDUW/graph.json","fetch_events":"https://pith.science/api/pith-number/BXI4CWW37Q2LBIXEJVYQKSWDUW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BXI4CWW37Q2LBIXEJVYQKSWDUW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BXI4CWW37Q2LBIXEJVYQKSWDUW/action/storage_attestation","attest_author":"https://pith.science/pith/BXI4CWW37Q2LBIXEJVYQKSWDUW/action/author_attestation","sign_citation":"https://pith.science/pith/BXI4CWW37Q2LBIXEJVYQKSWDUW/action/citation_signature","submit_replication":"https://pith.science/pith/BXI4CWW37Q2LBIXEJVYQKSWDUW/action/replication_record"}},"created_at":"2026-07-05T07:28:29.418623+00:00","updated_at":"2026-07-05T07:28:29.418623+00:00"}