{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LOFHDLRZCYO4XUQREUXI7KJZBF","short_pith_number":"pith:LOFHDLRZ","schema_version":"1.0","canonical_sha256":"5b8a71ae39161dcbd211252e8fa939097b97f678b8e6a3b07c6a7d5e63fc4e63","source":{"kind":"arxiv","id":"2507.05894","version":1},"attestation_state":"computed","paper":{"title":"MusiScene: Leveraging MU-LLaMA for Scene Imagination and Enhanced Video Background Music Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Fathinah Izzati, Gus Xia, Xinyue Li, Yuxuan Wu","submitted_at":"2025-07-08T11:32:02Z","abstract_excerpt":"Humans can imagine various atmospheres and settings when listening to music, envisioning movie scenes that complement each piece. For example, slow, melancholic music might evoke scenes of heartbreak, while upbeat melodies suggest celebration. This paper explores whether a Music Language Model, e.g. MU-LLaMA, can perform a similar task, called Music Scene Imagination (MSI), which requires cross-modal information from video and music to train. To improve upon existing music captioning models which focusing solely on musical elements, we introduce MusiScene, a music captioning model designed to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.05894","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-08T11:32:02Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ec2f655be689a18bd55d9be0991758851d7c726773b5cd77d1a895ab54a78a77","abstract_canon_sha256":"3cc571b5aac32a3b00390cadeeab64a9c332f19764221ac8aefb495042647176"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:33:37.901711Z","signature_b64":"ct55yb20Mrz9/Zrmv6FcsCVxaNwJIZkMVi1NawobT6bH2+Bw5RnJZX7hPJ0w8IXgYnY7AzrDeoCdSJHqMYe0CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b8a71ae39161dcbd211252e8fa939097b97f678b8e6a3b07c6a7d5e63fc4e63","last_reissued_at":"2026-07-05T11:33:37.901212Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:33:37.901212Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MusiScene: Leveraging MU-LLaMA for Scene Imagination and Enhanced Video Background Music Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Fathinah Izzati, Gus Xia, Xinyue Li, Yuxuan Wu","submitted_at":"2025-07-08T11:32:02Z","abstract_excerpt":"Humans can imagine various atmospheres and settings when listening to music, envisioning movie scenes that complement each piece. For example, slow, melancholic music might evoke scenes of heartbreak, while upbeat melodies suggest celebration. This paper explores whether a Music Language Model, e.g. MU-LLaMA, can perform a similar task, called Music Scene Imagination (MSI), which requires cross-modal information from video and music to train. To improve upon existing music captioning models which focusing solely on musical elements, we introduce MusiScene, a music captioning model designed to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.05894","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.05894/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.05894","created_at":"2026-07-05T11:33:37.901269+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.05894v1","created_at":"2026-07-05T11:33:37.901269+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.05894","created_at":"2026-07-05T11:33:37.901269+00:00"},{"alias_kind":"pith_short_12","alias_value":"LOFHDLRZCYO4","created_at":"2026-07-05T11:33:37.901269+00:00"},{"alias_kind":"pith_short_16","alias_value":"LOFHDLRZCYO4XUQR","created_at":"2026-07-05T11:33:37.901269+00:00"},{"alias_kind":"pith_short_8","alias_value":"LOFHDLRZ","created_at":"2026-07-05T11:33:37.901269+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LOFHDLRZCYO4XUQREUXI7KJZBF","json":"https://pith.science/pith/LOFHDLRZCYO4XUQREUXI7KJZBF.json","graph_json":"https://pith.science/api/pith-number/LOFHDLRZCYO4XUQREUXI7KJZBF/graph.json","events_json":"https://pith.science/api/pith-number/LOFHDLRZCYO4XUQREUXI7KJZBF/events.json","paper":"https://pith.science/paper/LOFHDLRZ"},"agent_actions":{"view_html":"https://pith.science/pith/LOFHDLRZCYO4XUQREUXI7KJZBF","download_json":"https://pith.science/pith/LOFHDLRZCYO4XUQREUXI7KJZBF.json","view_paper":"https://pith.science/paper/LOFHDLRZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.05894&json=true","fetch_graph":"https://pith.science/api/pith-number/LOFHDLRZCYO4XUQREUXI7KJZBF/graph.json","fetch_events":"https://pith.science/api/pith-number/LOFHDLRZCYO4XUQREUXI7KJZBF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LOFHDLRZCYO4XUQREUXI7KJZBF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LOFHDLRZCYO4XUQREUXI7KJZBF/action/storage_attestation","attest_author":"https://pith.science/pith/LOFHDLRZCYO4XUQREUXI7KJZBF/action/author_attestation","sign_citation":"https://pith.science/pith/LOFHDLRZCYO4XUQREUXI7KJZBF/action/citation_signature","submit_replication":"https://pith.science/pith/LOFHDLRZCYO4XUQREUXI7KJZBF/action/replication_record"}},"created_at":"2026-07-05T11:33:37.901269+00:00","updated_at":"2026-07-05T11:33:37.901269+00:00"}