{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OOJOZCXEPQOA4T7KE3GZPI7VM4","short_pith_number":"pith:OOJOZCXE","schema_version":"1.0","canonical_sha256":"7392ec8ae47c1c0e4fea26cd97a3f567232f221a30c7d8f5a3cc8542ac401bb8","source":{"kind":"arxiv","id":"2410.21943","version":1},"attestation_state":"computed","paper":{"title":"Beyond Text: Optimizing RAG with Multimodal Inputs for Industrial Applications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Monica Riedler, Stefan Langer","submitted_at":"2024-10-29T11:03:31Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated impressive capabilities in answering questions, but they lack domain-specific knowledge and are prone to hallucinations. Retrieval Augmented Generation (RAG) is one approach to address these challenges, while multimodal models are emerging as promising AI assistants for processing both text and images. In this paper we describe a series of experiments aimed at determining how to best integrate multimodal models into RAG systems for the industrial domain. The purpose of the experiments is to determine whether including images alongside text from do"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.21943","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-29T11:03:31Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"550157ff75a2fcec2b25f6db17ae180ade45511f5f5b3f2b463abbefeb823818","abstract_canon_sha256":"c9fb190071f8ef502f15a3eb2d171e72fac4efa3f848255248d5c24f068fad37"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:28:02.051654Z","signature_b64":"uQ2DvgWa7+ghR1uMZjLmPen3w+oEofzGHslkSiDQNuJCizl1Uv5baiZBZNt0K1G8QY9dYWZ2oOq4GGpeoYX1Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7392ec8ae47c1c0e4fea26cd97a3f567232f221a30c7d8f5a3cc8542ac401bb8","last_reissued_at":"2026-07-05T09:28:02.051165Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:28:02.051165Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Text: Optimizing RAG with Multimodal Inputs for Industrial Applications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Monica Riedler, Stefan Langer","submitted_at":"2024-10-29T11:03:31Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated impressive capabilities in answering questions, but they lack domain-specific knowledge and are prone to hallucinations. Retrieval Augmented Generation (RAG) is one approach to address these challenges, while multimodal models are emerging as promising AI assistants for processing both text and images. In this paper we describe a series of experiments aimed at determining how to best integrate multimodal models into RAG systems for the industrial domain. The purpose of the experiments is to determine whether including images alongside text from do"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.21943","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.21943/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.21943","created_at":"2026-07-05T09:28:02.051225+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.21943v1","created_at":"2026-07-05T09:28:02.051225+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.21943","created_at":"2026-07-05T09:28:02.051225+00:00"},{"alias_kind":"pith_short_12","alias_value":"OOJOZCXEPQOA","created_at":"2026-07-05T09:28:02.051225+00:00"},{"alias_kind":"pith_short_16","alias_value":"OOJOZCXEPQOA4T7K","created_at":"2026-07-05T09:28:02.051225+00:00"},{"alias_kind":"pith_short_8","alias_value":"OOJOZCXE","created_at":"2026-07-05T09:28:02.051225+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OOJOZCXEPQOA4T7KE3GZPI7VM4","json":"https://pith.science/pith/OOJOZCXEPQOA4T7KE3GZPI7VM4.json","graph_json":"https://pith.science/api/pith-number/OOJOZCXEPQOA4T7KE3GZPI7VM4/graph.json","events_json":"https://pith.science/api/pith-number/OOJOZCXEPQOA4T7KE3GZPI7VM4/events.json","paper":"https://pith.science/paper/OOJOZCXE"},"agent_actions":{"view_html":"https://pith.science/pith/OOJOZCXEPQOA4T7KE3GZPI7VM4","download_json":"https://pith.science/pith/OOJOZCXEPQOA4T7KE3GZPI7VM4.json","view_paper":"https://pith.science/paper/OOJOZCXE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.21943&json=true","fetch_graph":"https://pith.science/api/pith-number/OOJOZCXEPQOA4T7KE3GZPI7VM4/graph.json","fetch_events":"https://pith.science/api/pith-number/OOJOZCXEPQOA4T7KE3GZPI7VM4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OOJOZCXEPQOA4T7KE3GZPI7VM4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OOJOZCXEPQOA4T7KE3GZPI7VM4/action/storage_attestation","attest_author":"https://pith.science/pith/OOJOZCXEPQOA4T7KE3GZPI7VM4/action/author_attestation","sign_citation":"https://pith.science/pith/OOJOZCXEPQOA4T7KE3GZPI7VM4/action/citation_signature","submit_replication":"https://pith.science/pith/OOJOZCXEPQOA4T7KE3GZPI7VM4/action/replication_record"}},"created_at":"2026-07-05T09:28:02.051225+00:00","updated_at":"2026-07-05T09:28:02.051225+00:00"}