{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:E7CU3FFHLF73OTBDCJK2XQIFKE","short_pith_number":"pith:E7CU3FFH","schema_version":"1.0","canonical_sha256":"27c54d94a7597fb74c231255abc105511912471dd765d96c330047d1c215ed19","source":{"kind":"arxiv","id":"2408.03936","version":1},"attestation_state":"computed","paper":{"title":"SLIM-RAFT: A Novel Fine-Tuning Approach to Improve Cross-Linguistic Performance for Mercosur Common Nomenclature","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Li Weigang, Pedro Carvalho Brom, Victor Rafael R. Celestino, Vin\\'icius Di Oliveira, Yuri Fa\\c{c}anha Bezerra","submitted_at":"2024-08-07T17:54:21Z","abstract_excerpt":"Natural language processing (NLP) has seen significant advancements with the advent of large language models (LLMs). However, substantial improvements are still needed for languages other than English, especially for specific domains like the applications of Mercosur Common Nomenclature (NCM), a Brazilian Harmonized System (HS). To address this gap, this study uses TeenyTineLLaMA, a foundational Portuguese LLM, as an LLM source to implement the NCM application processing. Additionally, a simplified Retrieval-Augmented Fine-Tuning (RAFT) technique, termed SLIM-RAFT, is proposed for task-specifi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.03936","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-07T17:54:21Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e14414b6904450bb449fd00f4b310307c6761c122d643afc66d37dece5360cec","abstract_canon_sha256":"ed47f99730a9afba103a6edc63b99ac092a83dc3f7322c7642be81a0d1290b53"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:53:13.235418Z","signature_b64":"sYrPyGVjW4HQC89cmCKXHJXH2F7nVndVzqxahPfohXJQ9Sns+dZoopmGjyzztLgdWUQUebFpRbKN/Gfq8DxYDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27c54d94a7597fb74c231255abc105511912471dd765d96c330047d1c215ed19","last_reissued_at":"2026-07-05T08:53:13.235005Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:53:13.235005Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SLIM-RAFT: A Novel Fine-Tuning Approach to Improve Cross-Linguistic Performance for Mercosur Common Nomenclature","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Li Weigang, Pedro Carvalho Brom, Victor Rafael R. Celestino, Vin\\'icius Di Oliveira, Yuri Fa\\c{c}anha Bezerra","submitted_at":"2024-08-07T17:54:21Z","abstract_excerpt":"Natural language processing (NLP) has seen significant advancements with the advent of large language models (LLMs). However, substantial improvements are still needed for languages other than English, especially for specific domains like the applications of Mercosur Common Nomenclature (NCM), a Brazilian Harmonized System (HS). To address this gap, this study uses TeenyTineLLaMA, a foundational Portuguese LLM, as an LLM source to implement the NCM application processing. Additionally, a simplified Retrieval-Augmented Fine-Tuning (RAFT) technique, termed SLIM-RAFT, is proposed for task-specifi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.03936","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.03936/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.03936","created_at":"2026-07-05T08:53:13.235060+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.03936v1","created_at":"2026-07-05T08:53:13.235060+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.03936","created_at":"2026-07-05T08:53:13.235060+00:00"},{"alias_kind":"pith_short_12","alias_value":"E7CU3FFHLF73","created_at":"2026-07-05T08:53:13.235060+00:00"},{"alias_kind":"pith_short_16","alias_value":"E7CU3FFHLF73OTBD","created_at":"2026-07-05T08:53:13.235060+00:00"},{"alias_kind":"pith_short_8","alias_value":"E7CU3FFH","created_at":"2026-07-05T08:53:13.235060+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.08174","citing_title":"LLM-BT-Terms: Back-Translation as a Framework for Terminology Standardization and Dynamic Semantic Embedding","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E7CU3FFHLF73OTBDCJK2XQIFKE","json":"https://pith.science/pith/E7CU3FFHLF73OTBDCJK2XQIFKE.json","graph_json":"https://pith.science/api/pith-number/E7CU3FFHLF73OTBDCJK2XQIFKE/graph.json","events_json":"https://pith.science/api/pith-number/E7CU3FFHLF73OTBDCJK2XQIFKE/events.json","paper":"https://pith.science/paper/E7CU3FFH"},"agent_actions":{"view_html":"https://pith.science/pith/E7CU3FFHLF73OTBDCJK2XQIFKE","download_json":"https://pith.science/pith/E7CU3FFHLF73OTBDCJK2XQIFKE.json","view_paper":"https://pith.science/paper/E7CU3FFH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.03936&json=true","fetch_graph":"https://pith.science/api/pith-number/E7CU3FFHLF73OTBDCJK2XQIFKE/graph.json","fetch_events":"https://pith.science/api/pith-number/E7CU3FFHLF73OTBDCJK2XQIFKE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E7CU3FFHLF73OTBDCJK2XQIFKE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E7CU3FFHLF73OTBDCJK2XQIFKE/action/storage_attestation","attest_author":"https://pith.science/pith/E7CU3FFHLF73OTBDCJK2XQIFKE/action/author_attestation","sign_citation":"https://pith.science/pith/E7CU3FFHLF73OTBDCJK2XQIFKE/action/citation_signature","submit_replication":"https://pith.science/pith/E7CU3FFHLF73OTBDCJK2XQIFKE/action/replication_record"}},"created_at":"2026-07-05T08:53:13.235060+00:00","updated_at":"2026-07-05T08:53:13.235060+00:00"}