{"as_of":"2026-08-11T11:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:05954ff39fb1fbae3a73c325eb3f7ab9f9e9771c12feb62581fa04da4a674600","coverage":[{"denominator":67,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":67,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T18:23:31.599121Z","state":"measured"},{"denominator":67,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":67,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.00029/citation-record","integrity":"/paper/2509.00029/integrity","json":"/paper/2509.00029/citation-record.json","paper":"/paper/2509.00029"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:41.976909Z","title":"Secure & Personalized Music-to- Video Generation via CHARCHA, 2025","venue":null,"work_id":"dc79bb02-8fe3-4548-9cc9-45d684f230a0","year":2025},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:26.897893Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:2700325a9f8fe10357a63de12e5a977575795f7daedc9ac29d57177501acfb17","observation_id":"35b71881-cbfc-4364-bd1e-bc64100746e9","resolution":{"observed_at":"2026-08-05T18:23:42.054602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:41.795857Z","title":"ImproveYourVideos: Architectural Im- provements for Text-to-Video Generation Pipeline.IEEE Ac- cess, 13:1986–2003, 2025","venue":null,"work_id":"129f407f-5461-4582-9766-a48e657a97bb","year":1986},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:26.959917Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:b43ab447096a8d3811003177aa2d53dda3e6f7ed92ea03cc89d4bdbb3f7fd011","observation_id":"a61a3c3b-bbed-48d2-94ba-e2d358488d00","resolution":{"observed_at":"2026-08-05T18:23:41.898128Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:41.638651Z","title":"The MIT Press, Cambridge, Massachusetts, 2021","venue":null,"work_id":"704bf259-97ac-46bd-83d2-2eb764f10a09","year":2021},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.037973Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:8e3c544d486d2067f99840184ee500a9a0fddcb2a58954782ad9b848230b7dc0","observation_id":"e8302f5f-55e2-4c94-90e2-e384f004d410","resolution":{"observed_at":"2026-08-05T18:23:41.702044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:41.504015Z","title":"Boden and Ernest A","venue":null,"work_id":"640bb127-c51b-49cf-b221-6f86a126ee33","year":2009},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.112468Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:78b73a723d8c88b23f79d41c411fa27a0f29c4276bd6573e9009086f4cd0dd4a","observation_id":"5ba3e4db-b1de-4f28-aca2-68fa68d69f68","resolution":{"observed_at":"2026-08-05T18:23:41.545856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:41.326779Z","title":"Review of Gottschall (2012): The story- telling animal: How stories make us human.Scientific Study of Literature, 2(2):317–321, 2012","venue":null,"work_id":"19471070-ade7-4635-badc-fdb54aac5050","year":2012},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.193693Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:69a69828f7ce26a3e189b302dfcb483c93f10cab78e571fd028161fc8e972025","observation_id":"bdad2ee3-a30d-475a-b094-ded1e023e0cd","resolution":{"observed_at":"2026-08-05T18:23:41.416285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.11722","last_updated":"2023-05-31T16:02:39Z","snapshot_observed_at":"2026-08-07T09:48:52.816033Z","submitted_at":"2023-01-27T14:08:15Z","title":"Diffusion Models as Artists: Are we Closing the Gap between Humans and Machines?","version":3},"cited_work":{"arxiv_id":"2301.11722","doi":null,"metadata_source":"pith","pith_arxiv_id":"2301.11722","snapshot_observed_at":"2026-08-05T18:23:32.171695Z","title":"Diffusion Models as Artists: Are we Closing the Gap between Humans and Machines?","venue":"cs.AI","work_id":"b9939937-8602-44a5-b94d-508d041c34f6","year":2023},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.331853Z"},"links":{"cited_paper":"/paper/2301.11722","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:7ef654ea2094dddf0f5f9c8ce98b81f19c75b10b1e87f485892c5d43f8a08efc","observation_id":"c291cf6f-437e-4aab-912a-8393c68895a1","resolution":{"observed_at":"2026-08-05T18:23:32.266214Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:41.146383Z","title":"Crossmodal associations between naturally occurring tactile and sound textures.Per- ception, 53(4):219–239, 2024","venue":null,"work_id":"e91e5cb3-d130-4477-bbcf-783a72c17c2d","year":2024},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.415389Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:ebe24b3d89b10c434a4c71e0aa914e6d2412c0825acf9225cf380a3373c7cd02","observation_id":"4c861e84-251a-4ec6-a422-bb2a7f26388b","resolution":{"observed_at":"2026-08-05T18:23:41.268228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:40.967633Z","title":"Cancino-Chac ´on, Maarten Grachten, Werner Goebl, and Gerhard Widmer","venue":null,"work_id":"13bf2ce4-9b12-4086-a7a9-729092e51802","year":2018},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.487946Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:33050bec9ff3df3750f048ddfe8abf794cb54ba56ebdf3de60bf242082fb295d","observation_id":"1383f726-cf8c-4abb-b8f7-54389b6414c0","resolution":{"observed_at":"2026-08-05T18:23:41.039657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:40.809779Z","title":"”scary robots”: Examining public responses to ai","venue":null,"work_id":"2a13bb5c-1fe9-452e-8481-96b95347c28a","year":2019},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.566047Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:c12930aa2425c749a63dec1866d04f0e1bc6826c0c1a0869e307462ba43bc821","observation_id":"edca7429-49e1-4a90-b226-69ee5cb045ed","resolution":{"observed_at":"2026-08-05T18:23:40.868573Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2102.09109","last_updated":"2021-02-18T01:38:11Z","snapshot_observed_at":"2026-08-06T20:25:18.421198Z","submitted_at":"2021-02-18T01:38:11Z","title":"Understanding and Creating Art with AI: Review and Outlook","version":1},"cited_work":{"arxiv_id":"2102.09109","doi":null,"metadata_source":"pith","pith_arxiv_id":"2102.09109","snapshot_observed_at":"2026-08-05T18:23:32.006686Z","title":"Understanding and Creating Art with AI: Review and Outlook","venue":"cs.CV","work_id":"32925227-b784-4a7e-8ae2-680bc924b2d3","year":2021},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.698936Z"},"links":{"cited_paper":"/paper/2102.09109","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:273767e7e7981e9278769310993d3a2f57c693be4f4b2e9a29a5f6476cc47874","observation_id":"b0ca1abb-0212-4b0c-b9df-58d12a4c64f9","resolution":{"observed_at":"2026-08-05T18:23:32.063832Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:40.586373Z","title":"Dasovich-Wilson, Marc Thompson, and Suvi Saarikallio","venue":null,"work_id":"898bd4d9-7c02-4fff-9c52-6b1416f94e5f","year":2022},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.784957Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:ac7c91b43ddf7da8ec7aabddb04a0fd942d7e10d507324125be6e64293f55ec8","observation_id":"cebe88bd-f23b-4be7-8dd7-154a2ad1f1e6","resolution":{"observed_at":"2026-08-05T18:23:40.679610Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T18:23:27.845221Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.845221Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:7edb727011960000565c3066cb1ee5f5df32e49c7732ff8d119af30875565303","observation_id":"54998d99-14d2-4bad-bb6b-7cf9c35eff00","resolution":{"observed_at":"2026-08-05T18:23:27.845221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:40.420623Z","title":"Pengi: an audio language model for audio tasks","venue":null,"work_id":"1238dbbd-a3f3-4ce4-9a7f-2250452d6d0f","year":2023},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.892730Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:ddc4775550f49d66210bb410c4bd015c0e0d23437626038d236beca3ed15d79e","observation_id":"2b292f5d-f0b2-440a-b3b8-14ef5615f0c2","resolution":{"observed_at":"2026-08-05T18:23:40.508262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:40.248280Z","title":"Berkeley Publishing Group, New York, New York, 2005","venue":null,"work_id":"94a5ebd0-22d2-4d60-902d-55983c6a8119","year":2005},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:27.988577Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:15255cf8c86ac4d9b93837a7514450095d2c5337056710f3fe5ece906a55edc0","observation_id":"5438db99-ddb5-4b7e-b188-6f788828f549","resolution":{"observed_at":"2026-08-05T18:23:40.316719Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:40.031147Z","title":"EasyVid AI Video Maker, 2024","venue":null,"work_id":"33d759ae-69ea-4bcf-85c2-cca996fa6461","year":2024},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.068936Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:b7c571c71887df338bfdf7a748c6099f43fea5f3608e4542998c477920322067","observation_id":"ec55e3db-b401-425e-a0c1-e5f396de7d9e","resolution":{"observed_at":"2026-08-05T18:23:40.146291Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1706.07068","last_updated":"2017-06-21T18:05:13Z","snapshot_observed_at":"2026-07-06T05:47:54.334362Z","submitted_at":"2017-06-21T18:05:13Z","title":"CAN: Creative Adversarial Networks, Generating \"Art\" by Learning About Styles and Deviating from Style Norms","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.07068","snapshot_observed_at":"2026-08-05T18:23:28.167494Z","title":"CAN: Creative Adversarial Networks, Generating ”Art” by Learning About Styles and Deviating from Style Norms, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.167494Z"},"links":{"cited_paper":"/paper/1706.07068","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:6e34a617ef8e72f3432fa5ee0c80848adde2859329ec02ad92658085b1a1f978","observation_id":"f7ef522e-3c84-4be7-90a0-ec5035bf86e8","resolution":{"observed_at":"2026-08-05T18:23:28.167494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:39.806858Z","title":"CLAP Learning Audio Concepts from Natural Language Supervision.ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 1–5, 2023","venue":null,"work_id":"a152e43c-8f47-445c-83ac-b38b0499db8f","year":2023},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.237066Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:8d174f7ca090a474c543026dde37e8118bcf010c806610c6c942bef91e20453e","observation_id":"690b92dd-d1ba-415f-b0e0-47f770d65da6","resolution":{"observed_at":"2026-08-05T18:23:39.930859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:39.529096Z","title":"Rand, and Iyad Rah- wan","venue":null,"work_id":"01072721-bff2-4ead-b587-74dbcbc04113","year":2020},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.330599Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:7b81595412dcc30c26eb0685f43a10c8f32061380058328a7de102860859a926","observation_id":"d0d87f1c-8fa1-40e8-9acc-a015489e44bd","resolution":{"observed_at":"2026-08-05T18:23:39.659126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:39.223711Z","title":"Frank, Matthew Groh, Laura Herman, Neil Leach, Robert Mahari, Alex “Sandy” Pentland, Olga Russakovsky, Hope Schroeder, and Amy Smith","venue":null,"work_id":"fe001ab7-a92c-4f56-b34c-7f1d5c51edcf","year":2023},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.455570Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:811df81a2d50da12d8d51b7583834f339def257f035e0c2b8cc5a00804607112","observation_id":"28a48ced-bc6f-4edc-a4ae-bd6ef9e7845e","resolution":{"observed_at":"2026-08-05T18:23:39.392080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T18:23:28.522442Z","title":"Sakshi, Oriol Ni- eto, Ramani Duraiswami, and Dinesh Manocha","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.522442Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:f79668e3ce1b39af49015366d2cf1c593fccbbe714cfdc01f9b03a44b844337c","observation_id":"b03fa92e-d669-4b4c-b3cc-618d47076bdd","resolution":{"observed_at":"2026-08-05T18:23:28.522442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:38.916090Z","title":"From Ragtime to Swingtime: Fifty Glittering Years of Stage and Song","venue":null,"work_id":"c996553b-b423-49cf-86af-ee4941237ad0","year":1939},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.614731Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:5a4ac33f3e1542a1d1903f67ab263a2f62d25ecaed3adb81e7e125b7247c4738","observation_id":"6a5f500d-6737-4254-a0b2-d12bd44783da","resolution":{"observed_at":"2026-08-05T18:23:39.058349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:28.676982Z","title":"Generative adversarial networks.Commu- nications of the ACM, 63(11):139–144, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.676982Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:25bbe39e96c145ce05423a820abb346eb7d64d12e2493dcd2a4c07d500562502","observation_id":"cb10c913-26cb-4406-ba05-edefb477c907","resolution":{"observed_at":"2026-08-05T18:23:28.676982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:38.704976Z","title":"Graesser, Murray Singer, and Tom Trabasso","venue":null,"work_id":"be9f1da1-5a16-490b-849f-5ee5bfe2733d","year":1994},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.760858Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:92f907817509b3fddf4635ae01fdbf31fc8258dc712f0b208e11c77c460db015","observation_id":"d5233c9e-2fff-4183-9f49-a040759ab52c","resolution":{"observed_at":"2026-08-05T18:23:38.781115Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:38.540680Z","title":"Beware of fictional ai narratives.Nature Machine Intelligence, 2(11):654–654, 2020","venue":null,"work_id":"bc5fdce7-53f7-4bfb-aae4-6cd1e56828cc","year":2020},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.852552Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:0c91e62345065042d5c19425350fc8792356e1c9955a3b2d5487b4675956b97c","observation_id":"8e3b659f-8a81-491c-9cbb-9245ba06337e","resolution":{"observed_at":"2026-08-05T18:23:38.617464Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1801.04486","last_updated":"2018-05-08T03:45:56Z","snapshot_observed_at":"2026-07-06T06:18:30.390364Z","submitted_at":"2018-01-13T21:04:13Z","title":"Can Computers Create Art?","version":6},"cited_work":{"arxiv_id":"1801.04486","doi":null,"metadata_source":"pith","pith_arxiv_id":"1801.04486","snapshot_observed_at":"2026-08-05T18:23:31.807825Z","title":"Can Computers Create Art?","venue":"cs.AI","work_id":"cc98d60e-73a7-43e1-815d-3bdfe0ac3145","year":2018},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.947855Z"},"links":{"cited_paper":"/paper/1801.04486","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:db71170dce912a7c08552b1b787b817977410766a8bc28d4ad730da6dd3ad83b","observation_id":"c7a6a6e8-13fa-4429-99e4-29a1ce06a8b9","resolution":{"observed_at":"2026-08-05T18:23:31.876234Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:38.375127Z","title":"Computers do not make art, people do","venue":null,"work_id":"fe53bccf-b37b-4458-8ae7-099e007bd577","year":2020},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.024474Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:df90478fa8eb9b38ff831936e055a6838abcf97b0d07c9ce200a5c1faa9dc4c6","observation_id":"88c7f4d6-e73e-4761-89ed-973b8d6cd02f","resolution":{"observed_at":"2026-08-05T18:23:38.444000Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.15282","last_updated":"2021-12-17T17:21:04Z","snapshot_observed_at":"2026-08-07T06:41:10.116700Z","submitted_at":"2021-05-30T17:14:52Z","title":"Cascaded Diffusion Models for High Fidelity Image Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.15282","snapshot_observed_at":"2026-08-05T18:23:29.137643Z","title":"Fleet, Mohammad Norouzi, and Tim Salimans","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.137643Z"},"links":{"cited_paper":"/paper/2106.15282","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:5304a35f1986df68a939683a6351cfa7c8e1416195399b77d021cb781a0d1966","observation_id":"8474ac28-c3e5-4463-9bb9-edf5a15669f1","resolution":{"observed_at":"2026-08-05T18:23:29.137643Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.03458","last_updated":"2022-06-22T20:22:29Z","snapshot_observed_at":"2026-07-06T12:57:51.108636Z","submitted_at":"2022-04-07T14:08:02Z","title":"Video Diffusion Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.03458","snapshot_observed_at":"2026-08-05T18:23:29.193146Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.193146Z"},"links":{"cited_paper":"/paper/2204.03458","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:db467c60b12a7d6b00ab135648779e0c006f9c33f54d79deb75f4df37c47e9ce","observation_id":"2ece083d-74c1-4451-ab72-1d3abc6687e3","resolution":{"observed_at":"2026-08-05T18:23:29.193146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:38.174722Z","title":"Artificial Intelli- gence, Artists, and Art: Attitudes Toward Artwork Produced by Humans vs","venue":null,"work_id":"35b2de33-260c-4d80-a463-abd05d484366","year":2019},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.276918Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:040c12b7957a3aa660a3e622af2b474f9498bd85a1544b15889b242d59408f25","observation_id":"4240c902-f576-46d6-a20b-3c28a669161d","resolution":{"observed_at":"2026-08-05T18:23:38.287509Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:37.983000Z","title":"Blaine Horton Jr, Michael W","venue":null,"work_id":"dd0307df-c167-4daf-828d-6ef0b5a20fd5","year":2023},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.328957Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:93b8eca08fc18ff886e39bf98144a6016378b9664f1a408028fd90568ac60fe4","observation_id":"503c4c33-4102-42e1-8021-e4b0c846cba7","resolution":{"observed_at":"2026-08-05T18:23:38.054367Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:29.398758Z","title":"VBench: Com- prehensive Benchmark Suite for Video Generative Models,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.398758Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:7caf736260720f65413ad542fef30250ed3b93456091b3c0b572a1796298b597","observation_id":"41388663-6713-471c-ac63-3afa9d8e79ef","resolution":{"observed_at":"2026-08-05T18:23:29.398758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:37.763908Z","title":"Cross-modal associations between paintings and sounds: Ef- fects of embodiment.Perception, 51(12):871–888, 2022","venue":null,"work_id":"aabea5c7-ebfc-46ae-96c1-5b13b393199f","year":2022},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.573162Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:a024fa40df3e7bbc8d32af3590f6eba39cd08a6b3607fc36666825083b56a3c6","observation_id":"6b1ec288-881f-4ad2-baf4-f984b31c6e91","resolution":{"observed_at":"2026-08-05T18:23:37.848811Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:37.567997Z","title":"Kaiber AI: Generating Videos with Superstu- dio, 2025","venue":null,"work_id":"b5e11e54-974d-4b1e-921e-206e3aecce30","year":2025},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.639115Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:4b10e87b36f7b889e832e1f6891b847e17b452498dc5528b58aef219eeea9cf9","observation_id":"15bb7bbc-1eac-47d0-83b2-3fd9b6249836","resolution":{"observed_at":"2026-08-05T18:23:37.644475Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:37.389389Z","title":"Artificial Intelligence and Copyright: Le- gal Quandary in the Digital Age: Some Musings.SSRN Elec- tronic Journal, 2021","venue":null,"work_id":"4a26ff41-7480-4a10-9906-28851aedac6f","year":2021},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.678001Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:0559373eb859f3f2741bb841b1f78a3e019548c37b0252e89716d05f08c675a9","observation_id":"8704e49a-128b-44ba-a303-d5f9636c681a","resolution":{"observed_at":"2026-08-05T18:23:37.465831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:37.187489Z","title":"”AI enhances our performance, I have no doubt this one will do the same”: The Placebo effect is ro- bust to negative descriptions of AI","venue":null,"work_id":"9a05df86-de93-4629-9f56-a78b7e0bce41","year":2024},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.734159Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:ef3589beb8a5955314ed7b71ec945826a2e0e768ef03a0a18108236b93e6f1d0","observation_id":"698aae6f-f86f-4f8e-b48d-636c01161c93","resolution":{"observed_at":"2026-08-05T18:23:37.288366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:36.981079Z","title":"The (R)evolution of Music Video in American Music Industry.New Horizons in English Stud- ies, 8:163–176, 2023","venue":null,"work_id":"f83a74c9-ce0a-47e6-a83c-15e6ba6ded47","year":2023},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.780560Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:012d0aed9706bed24881f2f4d6a334be9145c43f25dfc6a1e9a809f909d138c5","observation_id":"5fa8b5dc-de0b-45a2-b11c-af5887acc0f9","resolution":{"observed_at":"2026-08-05T18:23:37.075736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:36.733188Z","title":"The Placebo Effect of Artificial Intelli- gence in Human–Computer Interaction.ACM Transactions on Computer-Human Interaction, 29(6):1–32, 2022","venue":null,"work_id":"5d4a2ec1-6bb3-413f-b22e-118ac0b715b8","year":2022},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.812172Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:2574e04d6688d9069ce7e7fae08e7fe94758556931e626b819bc7b018a69cb8c","observation_id":"022a4c7f-7d85-4e6e-9fd6-3b600fcd24f6","resolution":{"observed_at":"2026-08-05T18:23:36.853711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:36.508728Z","title":"Robust One Shot Audio to Video Generation","venue":null,"work_id":"c91df561-b59a-4ddc-a438-29e88f2d0f68","year":2020},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.912793Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:e9b6beb79b5d0f572123492da2cf22ba4643f681b2ddefb381973af1bfc4fbf7","observation_id":"4be64de3-089b-4070-8776-a39526b934c4","resolution":{"observed_at":"2026-08-05T18:23:36.615800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:36.315178Z","title":null,"venue":null,"work_id":"b38c95f6-ab58-40c7-83fa-f71e1d97d819","year":2025},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.975981Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:da5bc2d7a857ad4e1cda6a0c5d1012a78c1d9845c895ba1f805d82f2c44f934b","observation_id":"74bd9a09-60a2-4fcf-95e4-2f5fcce6270f","resolution":{"observed_at":"2026-08-05T18:23:36.408063Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:36.097074Z","title":"Lima, Carlos G","venue":null,"work_id":"1e94319e-b6fe-4cd2-900c-6dcfe650a3f7","year":2022},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.043470Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:5a3dea087e667d3e6ebfde7d714e136cd74202abcbcf6e0c1ef751f027dc8bfc","observation_id":"3d797bd9-75fa-468f-a6c2-3c4459d4bd54","resolution":{"observed_at":"2026-08-05T18:23:36.189412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:35.883393Z","title":"In ai we trust? effects of agency locus and transparency on uncertainty reduction in human–ai interac- tion.Journal of Computer-Mediated Communication, 26(6): 384–402, 2021","venue":null,"work_id":"5c024efc-20b0-4d27-8c54-8a0483ae686e","year":2021},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.099710Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:92337aa709c7869672c43f3bab5efe0074bb3e961ec6211be6f6cb7dad6a2661","observation_id":"ac8a5eb0-6a88-4dae-92d6-665e84ccc9e7","resolution":{"observed_at":"2026-08-05T18:23:35.961495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:35.714917Z","title":"Revisiting the Gold Standard: Grounding Summarization Evaluation with Robust Human Evaluation","venue":null,"work_id":"f770d1e7-3be1-4dfd-ab09-a32199c04793","year":2023},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.149947Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:d56c1d79cb3916627b602b4f78c4f467c7adff879797b16ce822a5a9ba48c945","observation_id":"3230a59a-0e88-4646-b8e0-6bce11a5b60c","resolution":{"observed_at":"2026-08-05T18:23:35.771559Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.01809","last_updated":"2024-07-15T12:21:56Z","snapshot_observed_at":"2026-08-10T16:51:23.461337Z","submitted_at":"2023-09-04T20:54:11Z","title":"Are Emergent Abilities in Large Language Models just In-Context Learning?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.01809","snapshot_observed_at":"2026-08-05T18:23:30.219156Z","title":"Are Emergent Abilities in Large Language Models just In-Context Learning?, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.219156Z"},"links":{"cited_paper":"/paper/2309.01809","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:bcadf04d7c0f902f19ab1454d51d4009d8a5e728bd20bc904b6cc3b132da6805","observation_id":"f95758a0-0e62-46ba-b3df-1982fbfe6de0","resolution":{"observed_at":"2026-08-05T18:23:30.219156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:35.517717Z","title":"Art, Creativity, and the Potential of Artificial Intelligence.Arts, 8(1):26, 2019","venue":null,"work_id":"18733266-40c0-4075-8370-d74355f367f2","year":2019},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.292545Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:1e3d7f2b9b4a6c25909abfdc09bae568e3d8c465fb9e9c6b511d944f04f8805a","observation_id":"00238615-afab-462f-8c81-689dc8f2f935","resolution":{"observed_at":"2026-08-05T18:23:35.597409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:35.339597Z","title":"Windows Media Player, 2025","venue":null,"work_id":"2b022673-c7d1-47e4-a1b2-ba580a59ef9a","year":2025},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.365220Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:bf74730c445d441776d12042d3ecdae2dfb2fd6578dd1f09f1cf0b1b2cad353a","observation_id":"538e82b9-738e-4623-80bd-8ac68965494d","resolution":{"observed_at":"2026-08-05T18:23:35.424087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:35.173702Z","title":"Com- putational Music Structure Analysis (Dagstuhl Seminar 16092)","venue":null,"work_id":"815834fc-52cd-49ab-a7c2-8dc50c0215f1","year":2016},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.413253Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:a588e8c451f8051a2d23d29d25515194d41388c763f06279e644ce1c5dc32f69","observation_id":"d9d86d38-155b-4fab-9b62-ab2753341979","resolution":{"observed_at":"2026-08-05T18:23:35.253847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:35.009680Z","title":"AI Music Video Generator, 2025","venue":null,"work_id":"e6ac96d2-5053-4c4b-b666-17399acdf8c8","year":2025},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.475329Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:c17287deb199a4f41984b29c3a865a065a19c73efd39039ee72e0c65c710ee41","observation_id":"8439a621-4d74-40e0-b9f8-c498a573d51c","resolution":{"observed_at":"2026-08-05T18:23:35.081445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:34.787855Z","title":"Hello GPT-4o by OpenAI, 2025","venue":null,"work_id":"0ee93bba-ca12-4ab0-8442-7cbc9dccf20a","year":2025},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.514243Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:3706e323a6d6bfbf2a353a4fdf0ebbcf21489b465e72915b2e724a12cfc760bf","observation_id":"4b2417f8-7576-4bb6-a7bd-c4c2be20e346","resolution":{"observed_at":"2026-08-05T18:23:34.888134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:34.549933Z","title":"Synaesthesia.European Neurology, 57(2):120– 124, 2007","venue":null,"work_id":"3fc1f95b-cbb1-4c9f-bd15-a9e1897ee65f","year":2007},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.606125Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:c94671d0c7fa7a7714d4b757039ecc1d59bdd5b6504e9bf8225a15ddf5bfe599","observation_id":"d06b6828-bac5-48b9-92cf-986da4e8dcf0","resolution":{"observed_at":"2026-08-05T18:23:34.657679Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:34.425635Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision,","venue":null,"work_id":"afc93f26-eef7-4af8-a9da-ded3075b7acb","year":null},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.667108Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:4134c84a3109598ebda7d47377dd34e4e25e7639594b2e3cad2d8f20e08133b8","observation_id":"4476f215-5682-4707-a602-e679e3427c66","resolution":{"observed_at":"2026-08-05T18:23:34.467417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:34.286731Z","title":"Self-supervised Dance Video Synthesis Conditioned on Mu- sic","venue":null,"work_id":"b31b37d1-fa15-46e7-9c8a-1def205db04d","year":null},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.785288Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:2891727aaf7ca0e1c7f6011a0add9e9623a2828a44787bbafb43f0f950e0864d","observation_id":"1e2aeb90-a502-4ce8-9300-b14926c3376d","resolution":{"observed_at":"2026-08-05T18:23:34.356357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:34.122366Z","title":"High-Resolution Image Synthesis With Latent Diffusion Models","venue":null,"work_id":"a6b91f29-baac-4296-bf7a-7d9777184ac9","year":2022},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.911969Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:e502752d94b64687b4a1599a8bc3cc3f61c98f9e57bbe20525f137440b285ef6","observation_id":"932469f3-9384-422f-b1db-2cbf574f3cf6","resolution":{"observed_at":"2026-08-05T18:23:34.209998Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:33.954837Z","title":"Oxford University PressNew York, NY , 1995","venue":null,"work_id":"40561e49-e113-4b8d-a696-546d96df1db8","year":1995},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.966040Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:171aad38114b33eec93459f77ec7087c2c7ab531a599c1960379ad4a44a8a617","observation_id":"c9259e24-855a-4884-84ad-ea6c0f4f8482","resolution":{"observed_at":"2026-08-05T18:23:34.033741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:33.801447Z","title":"Machine Learning Processes As Sources of Ambiguity: Insights from AI Art","venue":null,"work_id":"8c0e7159-995f-411d-a124-75af12f86a55","year":2024},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:31.056150Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:7eb34a2d3e7da84d76ccf15410d6e6afee498151761cd8c9421e836c09e198d3","observation_id":"329a479d-5675-4df6-b4d0-b612c99fed59","resolution":{"observed_at":"2026-08-05T18:23:33.859082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:33.684872Z","title":"Mochi 1 by Genmo Team, 2024","venue":null,"work_id":"d0510775-860b-40e6-a6f2-7177070da622","year":2024},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:31.097961Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:b7b00d1870e092336b2d8a440c959cdbbe21070a5b9f5cb007a0e32b945be41b","observation_id":"26ef0629-2ced-490d-b627-9b126447c808","resolution":{"observed_at":"2026-08-05T18:23:33.748561Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:33.547702Z","title":"Revid.ai, 2025","venue":null,"work_id":"ada854bf-9a3c-4cb1-882f-5a81809249ab","year":2025},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:31.151817Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:c56a1a8a88686ea9d0b983ea426bc2ad894bd8365af881d667d14608ebe597f0","observation_id":"6efd3b9f-d5b5-4315-8054-5d2f427bca2f","resolution":{"observed_at":"2026-08-05T18:23:33.639793Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:33.324723Z","title":"Specterr: Music Video Maker Online, 2025","venue":null,"work_id":"d36719c8-73f3-408f-83e9-c07086062af1","year":2025},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:31.179963Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:8720df46bdbefd974c76f1e2da508d2f244402767bcf54d661aff87c34007a59","observation_id":"14680611-1f71-4210-b9cf-19d4446e1bcf","resolution":{"observed_at":"2026-08-05T18:23:33.425690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20314","last_updated":"2025-04-19T02:22:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-26T08:25:43Z","title":"Wan: Open and Advanced Large-Scale Video Generative Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20314","snapshot_observed_at":"2026-08-05T18:23:31.222822Z","title":"Wan: Open and Advanced Large-Scale Video Generative Models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:31.222822Z"},"links":{"cited_paper":"/paper/2503.20314","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:e22cf09506d4430892295a26a236627397832794d4ea364644414b77eff52728","observation_id":"e2ae8bdf-dd3f-46a0-813c-da86bef81a7f","resolution":{"observed_at":"2026-08-05T18:23:31.222822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11903","last_updated":"2023-01-10T23:07:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-28T02:33:07Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.11903","snapshot_observed_at":"2026-08-05T18:23:31.278343Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:31.278343Z"},"links":{"cited_paper":"/paper/2201.11903","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:e8db84af0e52d3cb26d06e78e0ae2731112060889c0b45ab40088995eed0175a","observation_id":"41e18c0c-b122-424c-b254-b67062be121a","resolution":{"observed_at":"2026-08-05T18:23:31.278343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13116","last_updated":"2024-10-21T16:22:33Z","snapshot_observed_at":"2026-08-10T17:26:33.432994Z","submitted_at":"2024-02-20T16:17:37Z","title":"A Survey on Knowledge Distillation of Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13116","snapshot_observed_at":"2026-08-05T18:23:31.358582Z","title":"A Survey on Knowledge Distillation of Large Language Models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:31.358582Z"},"links":{"cited_paper":"/paper/2402.13116","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:525d1311931a8cbf6f94618a1e842beaa40c0fe25c42d75aba52d3752e3d33ab","observation_id":"e7073e7c-785f-4103-891a-2cb15cfe1d07","resolution":{"observed_at":"2026-08-05T18:23:31.358582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:33.140697Z","title":"Wordcraft: Story Writing With Large Language Models","venue":null,"work_id":"eb0aefe0-cfe3-4406-903f-a738f3df31f6","year":2022},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:31.419333Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:8354bc816878085776b4ee5c8122e08e6ea7044299e015a73bf6809835a3526f","observation_id":"49183d08-19ef-4db6-8e03-37ab707352bf","resolution":{"observed_at":"2026-08-05T18:23:33.201784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:32.860312Z","title":"Let’s Play Music: Audio-Driven Performance Video Generation","venue":null,"work_id":"c6837ca4-8a31-41e2-8106-030996e4c0b8","year":2021},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:31.483700Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:37a6cc10ed8597907a8867236e20d389acf0115e9390c5c002adbf98cec6cbd6","observation_id":"9634ef36-91ee-42ab-a47c-4f8eb2540052","resolution":{"observed_at":"2026-08-05T18:23:33.006421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:32.635037Z","title":"SCENE #:","venue":null,"work_id":"70119c26-2505-4f9f-805d-e8e58eb1baa2","year":null},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:31.551008Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:575f1a4bc1c651c1c1d2fabe9596a16277ffc9bec2985b447f55013f77919794","observation_id":"298bad70-e43a-43d4-9bcb-25f33bafe350","resolution":{"observed_at":"2026-08-05T18:23:32.741767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:32.400666Z","title":"All items were rated on a 7-point Likert scale: where 1 = Strongly Disagree, and 7 = Strongly Agree","venue":null,"work_id":"b13e6150-9762-4c9f-aa52-bd9a3a43c513","year":null},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:31.599121Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:67b73ddbdd0dc26b06de4ef179317829bb9364566a59d72fca21dee0435f561e","observation_id":"49386794-f5d9-4d91-b458-d8ee3e6eee88","resolution":{"observed_at":"2026-08-05T18:23:32.510045Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T18:23:30.852381Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.852381Z"},"links":{"citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:dc0dc7a00563730f8c01c43637165d01a09993c5ca3850aa2a7d6047fda1063e","observation_id":"93003541-2ae2-4dd4-8393-3136a5382996","resolution":{"observed_at":"2026-08-05T18:23:30.852381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.04356","last_updated":"2022-12-06T18:46:04Z","snapshot_observed_at":"2026-07-06T14:28:21.844826Z","submitted_at":"2022-12-06T18:46:04Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.04356","snapshot_observed_at":"2026-08-05T18:23:30.736114Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:30.736114Z"},"links":{"cited_paper":"/paper/2212.04356","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:47ff8dc127806af917faf92cb10d0250bf95fb38e26ea16d45b24ffc2b8d83d0","observation_id":"c7a423f1-b732-419d-a460-ae3e23fc4ffa","resolution":{"observed_at":"2026-08-05T18:23:30.736114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17982","last_updated":"2023-11-29T18:39:01Z","snapshot_observed_at":"2026-07-06T16:54:37.940527Z","submitted_at":"2023-11-29T18:39:01Z","title":"VBench: Comprehensive Benchmark Suite for Video Generative Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17982","snapshot_observed_at":"2026-08-05T18:23:29.510645Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:29.510645Z"},"links":{"cited_paper":"/paper/2311.17982","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:2479fb96903542d29882b8a0df33847abe268e1d86d02d2f7f21d72174e1113c","observation_id":"24b92ba4-8eb7-4a3e-897e-9e3f3a89ca43","resolution":{"observed_at":"2026-08-05T18:23:29.510645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-08T11:08:56.983262Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos"},"reference_resolution":{"displayed":67,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":15,"verified_exact":3,"verified_fuzzy":49},"total_outbound_references":67},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 67 of 67 outbound references and 0 inbound Pith citation observations for arXiv:2509.00029."}