{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JSMNPHA4GVZ6AVJKMP4PMY7CYA","short_pith_number":"pith:JSMNPHA4","schema_version":"1.0","canonical_sha256":"4c98d79c1c3573e0552a63f8f663e2c00a7ad8bc38e4026d30a1b4d2c131a2c3","source":{"kind":"arxiv","id":"2307.16372","version":1},"attestation_state":"computed","paper":{"title":"LP-MusicCaps: LLM-Based Pseudo Music Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Jongpil Lee, Juhan Nam, Keunwoo Choi, Seungheon Doh","submitted_at":"2023-07-31T02:32:02Z","abstract_excerpt":"Automatic music captioning, which generates natural language descriptions for given music tracks, holds significant potential for enhancing the understanding and organization of large volumes of musical data. Despite its importance, researchers face challenges due to the costly and time-consuming collection process of existing music-language datasets, which are limited in size. To address this data scarcity issue, we propose the use of large language models (LLMs) to artificially generate the description sentences from large-scale tag datasets. This results in approximately 2.2M captions paire"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.16372","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2023-07-31T02:32:02Z","cross_cats_sorted":["cs.IR","cs.MM","eess.AS"],"title_canon_sha256":"e702d62a7cdd73fb70a5ae4fb9c05a48bad7e2031ac5191960842ae67dd5b643","abstract_canon_sha256":"bf1e4c523ab6da83acde98345304baaad340e54f8a58c8220acf101741eb0b04"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:36:02.647598Z","signature_b64":"BBrmzRO8tiUBnN9044xUn3uBjKWp4nWIToMNgYb+nfYEUM5R2+2z6/RxFzzSQM0EfT7nH2BKlArYC9R1bCiZBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c98d79c1c3573e0552a63f8f663e2c00a7ad8bc38e4026d30a1b4d2c131a2c3","last_reissued_at":"2026-07-05T06:36:02.647125Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:36:02.647125Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LP-MusicCaps: LLM-Based Pseudo Music Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Jongpil Lee, Juhan Nam, Keunwoo Choi, Seungheon Doh","submitted_at":"2023-07-31T02:32:02Z","abstract_excerpt":"Automatic music captioning, which generates natural language descriptions for given music tracks, holds significant potential for enhancing the understanding and organization of large volumes of musical data. Despite its importance, researchers face challenges due to the costly and time-consuming collection process of existing music-language datasets, which are limited in size. To address this data scarcity issue, we propose the use of large language models (LLMs) to artificially generate the description sentences from large-scale tag datasets. This results in approximately 2.2M captions paire"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.16372","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.16372/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.16372","created_at":"2026-07-05T06:36:02.647176+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.16372v1","created_at":"2026-07-05T06:36:02.647176+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.16372","created_at":"2026-07-05T06:36:02.647176+00:00"},{"alias_kind":"pith_short_12","alias_value":"JSMNPHA4GVZ6","created_at":"2026-07-05T06:36:02.647176+00:00"},{"alias_kind":"pith_short_16","alias_value":"JSMNPHA4GVZ6AVJK","created_at":"2026-07-05T06:36:02.647176+00:00"},{"alias_kind":"pith_short_8","alias_value":"JSMNPHA4","created_at":"2026-07-05T06:36:02.647176+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.05971","citing_title":"Multimodal Video-to-Music Recommendation via Semantic Retrieval and Temporal Reranking","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31338","citing_title":"Beyond Binary Instrument QA: Probing Instrument Grounding in Music Audio-Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25343","citing_title":"Toward Native Multimodal Modeling: A Roadmap","ref_index":167,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27838","citing_title":"Dasheng AudioGen: A Unified Model for Generating Coherent Audio Scenes from Text","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2412.03603","citing_title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2506.14148","citing_title":"Acoustic scattering AI for non-invasive object classifications: A case study on hair assessment","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2310.13289","citing_title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2309.05922","citing_title":"A Survey of Hallucination in Large Foundation Models","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2507.08128","citing_title":"Audio Flamingo 3: Advancing Audio Intelligence with Fully Open Large Audio Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13431","citing_title":"Text2Score: Generating Sheet Music From Textual Prompts","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09120","citing_title":"Reddit2Deezer: A Scalable Dataset for Real-World Grounded Conversational Music Recommendation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2503.20215","citing_title":"Qwen2.5-Omni Technical Report","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15849","citing_title":"TinyMU: A Compact Audio-Language Model for Music Understanding","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JSMNPHA4GVZ6AVJKMP4PMY7CYA","json":"https://pith.science/pith/JSMNPHA4GVZ6AVJKMP4PMY7CYA.json","graph_json":"https://pith.science/api/pith-number/JSMNPHA4GVZ6AVJKMP4PMY7CYA/graph.json","events_json":"https://pith.science/api/pith-number/JSMNPHA4GVZ6AVJKMP4PMY7CYA/events.json","paper":"https://pith.science/paper/JSMNPHA4"},"agent_actions":{"view_html":"https://pith.science/pith/JSMNPHA4GVZ6AVJKMP4PMY7CYA","download_json":"https://pith.science/pith/JSMNPHA4GVZ6AVJKMP4PMY7CYA.json","view_paper":"https://pith.science/paper/JSMNPHA4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.16372&json=true","fetch_graph":"https://pith.science/api/pith-number/JSMNPHA4GVZ6AVJKMP4PMY7CYA/graph.json","fetch_events":"https://pith.science/api/pith-number/JSMNPHA4GVZ6AVJKMP4PMY7CYA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JSMNPHA4GVZ6AVJKMP4PMY7CYA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JSMNPHA4GVZ6AVJKMP4PMY7CYA/action/storage_attestation","attest_author":"https://pith.science/pith/JSMNPHA4GVZ6AVJKMP4PMY7CYA/action/author_attestation","sign_citation":"https://pith.science/pith/JSMNPHA4GVZ6AVJKMP4PMY7CYA/action/citation_signature","submit_replication":"https://pith.science/pith/JSMNPHA4GVZ6AVJKMP4PMY7CYA/action/replication_record"}},"created_at":"2026-07-05T06:36:02.647176+00:00","updated_at":"2026-07-05T06:36:02.647176+00:00"}