{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PVWARV4K57JJ2TKJJ4XAG5VCBB","short_pith_number":"pith:PVWARV4K","schema_version":"1.0","canonical_sha256":"7d6c08d78aefd29d4d494f2e0376a20841256a1c28addd5f8a39c933a17f9654","source":{"kind":"arxiv","id":"2407.09209","version":2},"attestation_state":"computed","paper":{"title":"Pronunciation Assessment with Multi-modal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.CL","authors_text":"Kaiqi Fu, Linkai Peng, Nan Yang, Shuran Zhou","submitted_at":"2024-07-12T12:16:14Z","abstract_excerpt":"Large language models (LLMs), renowned for their powerful conversational abilities, are widely recognized as exceptional tools in the field of education, particularly in the context of automated intelligent instruction systems for language learning. In this paper, we propose a scoring system based on LLMs, motivated by their positive impact on text-related scoring tasks. Specifically, the speech encoder first maps the learner's speech into contextual features. The adapter layer then transforms these features to align with the text embedding in latent space. The assessment task-specific prefix "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.09209","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-12T12:16:14Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"31f71db566828a315e84d9295b2de2df301462d6408f5daf12d53db9103cc9b8","abstract_canon_sha256":"1c6e845d705ac1bb323d5535d252b29ba2d5d60b3789e10934aa201f68ecf52f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:30.805936Z","signature_b64":"Hp6Kbxls8wLpuDGZXGezf194zSqHfIgTLGS21fXqXFIaj7B+0B5Bfd5X+einduoK5GgIDeVxp/NjebYNsPL0Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7d6c08d78aefd29d4d494f2e0376a20841256a1c28addd5f8a39c933a17f9654","last_reissued_at":"2026-07-05T08:45:30.805541Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:30.805541Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pronunciation Assessment with Multi-modal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.CL","authors_text":"Kaiqi Fu, Linkai Peng, Nan Yang, Shuran Zhou","submitted_at":"2024-07-12T12:16:14Z","abstract_excerpt":"Large language models (LLMs), renowned for their powerful conversational abilities, are widely recognized as exceptional tools in the field of education, particularly in the context of automated intelligent instruction systems for language learning. In this paper, we propose a scoring system based on LLMs, motivated by their positive impact on text-related scoring tasks. Specifically, the speech encoder first maps the learner's speech into contextual features. The adapter layer then transforms these features to align with the text embedding in latent space. The assessment task-specific prefix "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.09209","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.09209/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.09209","created_at":"2026-07-05T08:45:30.805598+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.09209v2","created_at":"2026-07-05T08:45:30.805598+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.09209","created_at":"2026-07-05T08:45:30.805598+00:00"},{"alias_kind":"pith_short_12","alias_value":"PVWARV4K57JJ","created_at":"2026-07-05T08:45:30.805598+00:00"},{"alias_kind":"pith_short_16","alias_value":"PVWARV4K57JJ2TKJ","created_at":"2026-07-05T08:45:30.805598+00:00"},{"alias_kind":"pith_short_8","alias_value":"PVWARV4K","created_at":"2026-07-05T08:45:30.805598+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.02915","citing_title":"English Pronunciation Evaluation without Complex Joint Training: LoRA Fine-tuned Speech Multimodal LLM","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PVWARV4K57JJ2TKJJ4XAG5VCBB","json":"https://pith.science/pith/PVWARV4K57JJ2TKJJ4XAG5VCBB.json","graph_json":"https://pith.science/api/pith-number/PVWARV4K57JJ2TKJJ4XAG5VCBB/graph.json","events_json":"https://pith.science/api/pith-number/PVWARV4K57JJ2TKJJ4XAG5VCBB/events.json","paper":"https://pith.science/paper/PVWARV4K"},"agent_actions":{"view_html":"https://pith.science/pith/PVWARV4K57JJ2TKJJ4XAG5VCBB","download_json":"https://pith.science/pith/PVWARV4K57JJ2TKJJ4XAG5VCBB.json","view_paper":"https://pith.science/paper/PVWARV4K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.09209&json=true","fetch_graph":"https://pith.science/api/pith-number/PVWARV4K57JJ2TKJJ4XAG5VCBB/graph.json","fetch_events":"https://pith.science/api/pith-number/PVWARV4K57JJ2TKJJ4XAG5VCBB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PVWARV4K57JJ2TKJJ4XAG5VCBB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PVWARV4K57JJ2TKJJ4XAG5VCBB/action/storage_attestation","attest_author":"https://pith.science/pith/PVWARV4K57JJ2TKJJ4XAG5VCBB/action/author_attestation","sign_citation":"https://pith.science/pith/PVWARV4K57JJ2TKJJ4XAG5VCBB/action/citation_signature","submit_replication":"https://pith.science/pith/PVWARV4K57JJ2TKJJ4XAG5VCBB/action/replication_record"}},"created_at":"2026-07-05T08:45:30.805598+00:00","updated_at":"2026-07-05T08:45:30.805598+00:00"}