{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:L2SM7KLA34H5PDJDCGXEZVEII6","short_pith_number":"pith:L2SM7KLA","schema_version":"1.0","canonical_sha256":"5ea4cfa960df0fd78d2311ae4cd48847a99b250e06d20312044faaeee9b9c1e3","source":{"kind":"arxiv","id":"2512.14926","version":2},"attestation_state":"computed","paper":{"title":"Parameter Efficient Multimodal Instruction Tuning for Romanian Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dumitru-Clementin Cercel, George-Andrei Dima, R\\u{a}zvan-Alexandru Sm\\u{a}du","submitted_at":"2025-12-16T21:36:28Z","abstract_excerpt":"Focusing on low-resource languages is an essential step toward democratizing generative AI. In this work, we contribute to reducing the multimodal NLP resource gap for Romanian. We translate the widely known Flickr30K dataset into Romanian and further extend it for visual question answering by leveraging open-source LLMs. We demonstrate the usefulness of our datasets by fine-tuning open-source VLMs on Romanian visual question answering. We select VLMs from three widely used model families: LLaMA 3.2, LLaVA 1.6, and Qwen2. For fine-tuning, we employ the parameter-efficient LoRA method. Our mode"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2512.14926","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-12-16T21:36:28Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"0af27525ee6075cc55824028964778f7fb845382a2dd79058636512bcfb2548a","abstract_canon_sha256":"a78933be6551c6755f23fb3cc9b215ea220e80d3ae7782ea91772a6a0cb4cd7b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T02:18:31.756541Z","signature_b64":"Yu5GrU+MF2Bu8pfLcxWCYNRtNef6YyBtJR2bovbfVupVhZsZwzdYMku00L9uNSRIapvk9cy6fnwtxWpD4EW8Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ea4cfa960df0fd78d2311ae4cd48847a99b250e06d20312044faaeee9b9c1e3","last_reissued_at":"2026-07-07T02:18:31.755665Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T02:18:31.755665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Parameter Efficient Multimodal Instruction Tuning for Romanian Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dumitru-Clementin Cercel, George-Andrei Dima, R\\u{a}zvan-Alexandru Sm\\u{a}du","submitted_at":"2025-12-16T21:36:28Z","abstract_excerpt":"Focusing on low-resource languages is an essential step toward democratizing generative AI. In this work, we contribute to reducing the multimodal NLP resource gap for Romanian. We translate the widely known Flickr30K dataset into Romanian and further extend it for visual question answering by leveraging open-source LLMs. We demonstrate the usefulness of our datasets by fine-tuning open-source VLMs on Romanian visual question answering. We select VLMs from three widely used model families: LLaMA 3.2, LLaVA 1.6, and Qwen2. For fine-tuning, we employ the parameter-efficient LoRA method. Our mode"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2512.14926","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2512.14926/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2512.14926","created_at":"2026-07-07T02:18:31.755768+00:00"},{"alias_kind":"arxiv_version","alias_value":"2512.14926v2","created_at":"2026-07-07T02:18:31.755768+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.14926","created_at":"2026-07-07T02:18:31.755768+00:00"},{"alias_kind":"pith_short_12","alias_value":"L2SM7KLA34H5","created_at":"2026-07-07T02:18:31.755768+00:00"},{"alias_kind":"pith_short_16","alias_value":"L2SM7KLA34H5PDJD","created_at":"2026-07-07T02:18:31.755768+00:00"},{"alias_kind":"pith_short_8","alias_value":"L2SM7KLA","created_at":"2026-07-07T02:18:31.755768+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2605.31401","citing_title":"\"\\^{I}n\\c{t}elegi Rom\\^ane\\c{s}te?'' A Recipe for Romanian Vision-Language Models","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L2SM7KLA34H5PDJDCGXEZVEII6","json":"https://pith.science/pith/L2SM7KLA34H5PDJDCGXEZVEII6.json","graph_json":"https://pith.science/api/pith-number/L2SM7KLA34H5PDJDCGXEZVEII6/graph.json","events_json":"https://pith.science/api/pith-number/L2SM7KLA34H5PDJDCGXEZVEII6/events.json","paper":"https://pith.science/paper/L2SM7KLA"},"agent_actions":{"view_html":"https://pith.science/pith/L2SM7KLA34H5PDJDCGXEZVEII6","download_json":"https://pith.science/pith/L2SM7KLA34H5PDJDCGXEZVEII6.json","view_paper":"https://pith.science/paper/L2SM7KLA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2512.14926&json=true","fetch_graph":"https://pith.science/api/pith-number/L2SM7KLA34H5PDJDCGXEZVEII6/graph.json","fetch_events":"https://pith.science/api/pith-number/L2SM7KLA34H5PDJDCGXEZVEII6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L2SM7KLA34H5PDJDCGXEZVEII6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L2SM7KLA34H5PDJDCGXEZVEII6/action/storage_attestation","attest_author":"https://pith.science/pith/L2SM7KLA34H5PDJDCGXEZVEII6/action/author_attestation","sign_citation":"https://pith.science/pith/L2SM7KLA34H5PDJDCGXEZVEII6/action/citation_signature","submit_replication":"https://pith.science/pith/L2SM7KLA34H5PDJDCGXEZVEII6/action/replication_record"}},"created_at":"2026-07-07T02:18:31.755768+00:00","updated_at":"2026-07-07T02:18:31.755768+00:00"}