{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JPTZ7S7QYWKMP27D2TVUSDCBW6","short_pith_number":"pith:JPTZ7S7Q","schema_version":"1.0","canonical_sha256":"4be79fcbf0c594c7ebe3d4eb490c41b7b7946d4ef305a7e83028b407bab7454c","source":{"kind":"arxiv","id":"2412.08194","version":2},"attestation_state":"computed","paper":{"title":"Magneto: Combining Small and Large Language Models for Schema Matching","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DB","authors_text":"Aecio Santos, Eden Wu, Eduardo Pena, Juliana Freire, Yurong Liu","submitted_at":"2024-12-11T08:35:56Z","abstract_excerpt":"Recent advances in language models opened new opportunities to address complex schema matching tasks. Schema matching approaches have been proposed that demonstrate the usefulness of language models, but they have also uncovered important limitations: Small language models (SLMs) require training data (which can be both expensive and challenging to obtain), and large language models (LLMs) often incur high computational costs and must deal with constraints imposed by context windows. We present Magneto, a cost-effective and accurate solution for schema matching that combines the advantages of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.08194","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DB","submitted_at":"2024-12-11T08:35:56Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"5feb7056e3b7f11a2bef6654448b8622d89f944de03948df8c6a8eb912a43b63","abstract_canon_sha256":"a89d85a241cbdfa014f88275a8de73d9c13848f606fd30fccdeac425d9238c7a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:34.747075Z","signature_b64":"usO3XYtXelVSrgimbNhUZvK9c3CISYWCLXTk1sDMxnAm9VUCmXbcNRsVwycs2bmag3hUCyakzpfLwpfo/sd0Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4be79fcbf0c594c7ebe3d4eb490c41b7b7946d4ef305a7e83028b407bab7454c","last_reissued_at":"2026-07-05T11:22:34.746569Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:34.746569Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Magneto: Combining Small and Large Language Models for Schema Matching","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DB","authors_text":"Aecio Santos, Eden Wu, Eduardo Pena, Juliana Freire, Yurong Liu","submitted_at":"2024-12-11T08:35:56Z","abstract_excerpt":"Recent advances in language models opened new opportunities to address complex schema matching tasks. Schema matching approaches have been proposed that demonstrate the usefulness of language models, but they have also uncovered important limitations: Small language models (SLMs) require training data (which can be both expensive and challenging to obtain), and large language models (LLMs) often incur high computational costs and must deal with constraints imposed by context windows. We present Magneto, a cost-effective and accurate solution for schema matching that combines the advantages of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.08194","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.08194/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.08194","created_at":"2026-07-05T11:22:34.746625+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.08194v2","created_at":"2026-07-05T11:22:34.746625+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.08194","created_at":"2026-07-05T11:22:34.746625+00:00"},{"alias_kind":"pith_short_12","alias_value":"JPTZ7S7QYWKM","created_at":"2026-07-05T11:22:34.746625+00:00"},{"alias_kind":"pith_short_16","alias_value":"JPTZ7S7QYWKMP27D","created_at":"2026-07-05T11:22:34.746625+00:00"},{"alias_kind":"pith_short_8","alias_value":"JPTZ7S7Q","created_at":"2026-07-05T11:22:34.746625+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.20482","citing_title":"ConStruM: A Structure-Guided LLM Framework for Context-Aware Schema Matching","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JPTZ7S7QYWKMP27D2TVUSDCBW6","json":"https://pith.science/pith/JPTZ7S7QYWKMP27D2TVUSDCBW6.json","graph_json":"https://pith.science/api/pith-number/JPTZ7S7QYWKMP27D2TVUSDCBW6/graph.json","events_json":"https://pith.science/api/pith-number/JPTZ7S7QYWKMP27D2TVUSDCBW6/events.json","paper":"https://pith.science/paper/JPTZ7S7Q"},"agent_actions":{"view_html":"https://pith.science/pith/JPTZ7S7QYWKMP27D2TVUSDCBW6","download_json":"https://pith.science/pith/JPTZ7S7QYWKMP27D2TVUSDCBW6.json","view_paper":"https://pith.science/paper/JPTZ7S7Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.08194&json=true","fetch_graph":"https://pith.science/api/pith-number/JPTZ7S7QYWKMP27D2TVUSDCBW6/graph.json","fetch_events":"https://pith.science/api/pith-number/JPTZ7S7QYWKMP27D2TVUSDCBW6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JPTZ7S7QYWKMP27D2TVUSDCBW6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JPTZ7S7QYWKMP27D2TVUSDCBW6/action/storage_attestation","attest_author":"https://pith.science/pith/JPTZ7S7QYWKMP27D2TVUSDCBW6/action/author_attestation","sign_citation":"https://pith.science/pith/JPTZ7S7QYWKMP27D2TVUSDCBW6/action/citation_signature","submit_replication":"https://pith.science/pith/JPTZ7S7QYWKMP27D2TVUSDCBW6/action/replication_record"}},"created_at":"2026-07-05T11:22:34.746625+00:00","updated_at":"2026-07-05T11:22:34.746625+00:00"}