{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:E66NSEIY4UAUJKJDFF47A3T2E2","short_pith_number":"pith:E66NSEIY","schema_version":"1.0","canonical_sha256":"27bcd91118e50144a9232979f06e7a26b29e177397f4fd40e0b8d8d531cd4c21","source":{"kind":"arxiv","id":"2501.11496","version":2},"attestation_state":"computed","paper":{"title":"Generative AI and Large Language Models in Language Preservation: Opportunities and Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Vincent Koc","submitted_at":"2025-01-20T14:03:40Z","abstract_excerpt":"The global crisis of language endangerment meets a technological turning point as Generative AI (GenAI) and Large Language Models (LLMs) unlock new frontiers in automating corpus creation, transcription, translation, and tutoring. However, this promise is imperiled by fragmented practices and the critical lack of a methodology to navigate the fraught balance between LLM capabilities and the profound risks of data scarcity, cultural misappropriation, and ethical missteps. This paper introduces a novel analytical framework that systematically evaluates GenAI applications against language-specifi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.11496","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-01-20T14:03:40Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"46766fc217580fc889faee8fba0a5256164bda34b0d5679e65ebf781811b22c7","abstract_canon_sha256":"4ddcb63aca049331ba5d41a08fb3b772026a470f1db462666fc97b31990ba4c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:44.099054Z","signature_b64":"TiWW5i3Yx1VX7KpQrdZQ9N8tn7tIBFcSUINXcqKGIgepG3hD0lHiy5HfrmggMSWSGPN2QWX4TTuzXDyXM3tGAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27bcd91118e50144a9232979f06e7a26b29e177397f4fd40e0b8d8d531cd4c21","last_reissued_at":"2026-07-05T11:04:44.098562Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:44.098562Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generative AI and Large Language Models in Language Preservation: Opportunities and Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Vincent Koc","submitted_at":"2025-01-20T14:03:40Z","abstract_excerpt":"The global crisis of language endangerment meets a technological turning point as Generative AI (GenAI) and Large Language Models (LLMs) unlock new frontiers in automating corpus creation, transcription, translation, and tutoring. However, this promise is imperiled by fragmented practices and the critical lack of a methodology to navigate the fraught balance between LLM capabilities and the profound risks of data scarcity, cultural misappropriation, and ethical missteps. This paper introduces a novel analytical framework that systematically evaluates GenAI applications against language-specifi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.11496","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.11496/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.11496","created_at":"2026-07-05T11:04:44.098622+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.11496v2","created_at":"2026-07-05T11:04:44.098622+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.11496","created_at":"2026-07-05T11:04:44.098622+00:00"},{"alias_kind":"pith_short_12","alias_value":"E66NSEIY4UAU","created_at":"2026-07-05T11:04:44.098622+00:00"},{"alias_kind":"pith_short_16","alias_value":"E66NSEIY4UAUJKJD","created_at":"2026-07-05T11:04:44.098622+00:00"},{"alias_kind":"pith_short_8","alias_value":"E66NSEIY","created_at":"2026-07-05T11:04:44.098622+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.12058","citing_title":"Tiny QA Benchmark++: Ultra-Lightweight, Synthetic Multilingual Dataset Generation & Smoke-Tests for Continuous LLM Evaluation","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E66NSEIY4UAUJKJDFF47A3T2E2","json":"https://pith.science/pith/E66NSEIY4UAUJKJDFF47A3T2E2.json","graph_json":"https://pith.science/api/pith-number/E66NSEIY4UAUJKJDFF47A3T2E2/graph.json","events_json":"https://pith.science/api/pith-number/E66NSEIY4UAUJKJDFF47A3T2E2/events.json","paper":"https://pith.science/paper/E66NSEIY"},"agent_actions":{"view_html":"https://pith.science/pith/E66NSEIY4UAUJKJDFF47A3T2E2","download_json":"https://pith.science/pith/E66NSEIY4UAUJKJDFF47A3T2E2.json","view_paper":"https://pith.science/paper/E66NSEIY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.11496&json=true","fetch_graph":"https://pith.science/api/pith-number/E66NSEIY4UAUJKJDFF47A3T2E2/graph.json","fetch_events":"https://pith.science/api/pith-number/E66NSEIY4UAUJKJDFF47A3T2E2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E66NSEIY4UAUJKJDFF47A3T2E2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E66NSEIY4UAUJKJDFF47A3T2E2/action/storage_attestation","attest_author":"https://pith.science/pith/E66NSEIY4UAUJKJDFF47A3T2E2/action/author_attestation","sign_citation":"https://pith.science/pith/E66NSEIY4UAUJKJDFF47A3T2E2/action/citation_signature","submit_replication":"https://pith.science/pith/E66NSEIY4UAUJKJDFF47A3T2E2/action/replication_record"}},"created_at":"2026-07-05T11:04:44.098622+00:00","updated_at":"2026-07-05T11:04:44.098622+00:00"}