{"as_of":"2026-08-08T09:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:614b95bfd609fa7b86b3f414c7227cb318d705bab073d2397fc2c7538b5e35f2","coverage":[{"denominator":71,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":71,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:15:48.995779Z","state":"measured"},{"denominator":71,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":71,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.19535/citation-record","integrity":"/paper/2505.19535/integrity","json":"/paper/2505.19535/citation-record.json","paper":"/paper/2505.19535"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:58.842603Z","title":"Tune-a-video: One-shot tuning of image diffusion models for text-to-video generation,","venue":null,"work_id":"d971d2d6-49de-449c-baf5-a1dd7fa13567","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:43.651761Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:f31be32d1d9d746675d4d41c4b25e217fd5aa7b5ee39ca318e170f05a25ed243","observation_id":"49259069-4357-497b-af56-bb497360a7ed","resolution":{"observed_at":"2026-08-07T14:15:58.979116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:58.544759Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation,","venue":null,"work_id":"5c4c451b-86a9-43eb-ad04-795c1a76b7c0","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:43.742055Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:82fb491d646a1619936b0913e43c8a14339227eb2836d56d00b55cfcf162a8e6","observation_id":"d726bc0c-741b-47db-9faa-edd19846277f","resolution":{"observed_at":"2026-08-07T14:15:58.676070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:58.212805Z","title":"Text2video-zero: Text-to-image diffusion models are zero-shot video generators,","venue":null,"work_id":"60ec42af-9790-43e7-9f6b-91717cc95542","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:43.841821Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:35b4dc1488cbe68251d055b2f60d5b57dd8f45792d6cb7d3809d3b902ad672dc","observation_id":"2f1b4a46-ee54-44d6-9c4c-13ed27702cc6","resolution":{"observed_at":"2026-08-07T14:15:58.429716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:58.028487Z","title":"Ccedit: Creative and controllable video editing via diffusion models,","venue":null,"work_id":"6dda3ce7-8d16-46d3-aad6-347e61f1b06a","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:43.901242Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:6ecb9a1a51a08d2a0b75571178e1ce8d3a5f930121bc75101c1e2ad5ad21426e","observation_id":"60d1a271-6100-4bc2-8999-e9635db4082b","resolution":{"observed_at":"2026-08-07T14:15:58.111035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:57.784564Z","title":"Controlvideo: Training-free controllable text-to-video generation,","venue":null,"work_id":"b0bf88dc-8d21-4477-8e0e-39c095d205fe","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.024176Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:0146800da336119ba023b942204cdba049f679987b13ce875af44447610cb84a","observation_id":"fb830f27-c3ed-4215-ad0d-1187587f0dab","resolution":{"observed_at":"2026-08-07T14:15:57.919016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:57.627990Z","title":"Fatezero: Fusing attentions for zero-shot text-based video editing,","venue":null,"work_id":"d69b191d-cb4a-4fde-9538-da6089af33fa","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.195828Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:bda119c8fbe854f44ffbd0c9fa9890911222906553bfb7016a2943ce8c8c19c5","observation_id":"b9a50780-125f-42b2-988f-e6e5d402579e","resolution":{"observed_at":"2026-08-07T14:15:57.713443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:57.457694Z","title":"Flatten: optical flow-guided attention for consistent text-to-video editing,","venue":null,"work_id":"f66f63e5-96af-4510-9f9f-235c5bd2da11","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.339750Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:a2486770c30420dcd16c07001fafc28a073bfe46b080d34726e2c0fbad3562a0","observation_id":"05c9366b-5047-46ac-8824-75e3d23a5552","resolution":{"observed_at":"2026-08-07T14:15:57.555158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:57.242692Z","title":"Fresco: Spatial-temporal correspondence for zero-shot video translation,","venue":null,"work_id":"aac59014-0c7a-409a-8059-bee1c8775f98","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.443157Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:8b8930719e6d3b5dac772e3d12fce6d8ffb2b972cd6d78a843a538c14f1c9b8b","observation_id":"2941ad73-01ea-4f6a-b494-4538eea50770","resolution":{"observed_at":"2026-08-07T14:15:57.301203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:57.079280Z","title":"Pix2video: Video editing using image diffusion,","venue":null,"work_id":"9be8cf9e-f38e-4d48-8d25-2cd6dc3cf809","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.574580Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:fbb2924930f752d6a9c87b0c9de24cd3db6ef323e8a43ea7503a2aa7ece136eb","observation_id":"47e9861c-8a2d-4354-b017-20dbd203fab1","resolution":{"observed_at":"2026-08-07T14:15:57.154855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.878966Z","title":"Rave: Randomized noise shuffling for fast and consistent video editing with diffusion models,","venue":null,"work_id":"f40233b8-cb97-4ef0-90a4-142ee7f3f70e","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.666645Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:02a6c643b608e3cad075a5fa44656afe2fb5cb6357d34622e34a360e8a3eae3e","observation_id":"77254bda-3784-410c-9557-33efcdda1cfb","resolution":{"observed_at":"2026-08-07T14:15:56.972919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.770806Z","title":"Slicedit: Zero-shot video editing with text-to-image diffusion models using spatio-temporal slices,","venue":null,"work_id":"f8dc0995-2847-42b6-a894-6ad3a52c7f27","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.788333Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:88ea3e47a509b2e0a9bf88f5a6aa55abfc00ffe3d53c8fceb4244f34cd577089","observation_id":"6d9acb26-1789-4300-a8ff-33f813aa1edb","resolution":{"observed_at":"2026-08-07T14:15:56.824036Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.643642Z","title":"Zero-shot video editing using off-the-shelf image diffusion models,","venue":null,"work_id":"55e1f8b2-c913-4bb3-8b61-9097eae0499f","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.869181Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:c6dca8888c7843984fc07cd37a7be5ce3f4602df24f4e9c9a1932d2828b4782d","observation_id":"ad892360-db24-4767-b16a-1ac2c692ca94","resolution":{"observed_at":"2026-08-07T14:15:56.720375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.473058Z","title":"Video quality assessment: A comprehensive survey,","venue":null,"work_id":"b1409ab5-6de1-4390-b588-711952994e6b","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:44.964072Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:8a8a7a1db3c00a6c32225e530086904d682a35f913d82a7045ef6dbede72b3a2","observation_id":"76e4353c-1fb6-48f1-ab9d-c5f18c8268da","resolution":{"observed_at":"2026-08-07T14:15:56.573558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.347184Z","title":"Exploring video quality assessment on user generated contents from aesthetic and technical perspectives,","venue":null,"work_id":"be44e27f-a6d5-403a-863d-fb4878b3f519","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.062144Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:24e5497884b67ae8303becf3ba5534bb00faff7393ffdd5bec4b774ed743b9d6","observation_id":"e4796152-584e-4f9a-81d4-47c9342fe8d8","resolution":{"observed_at":"2026-08-07T14:15:56.401381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:56.222297Z","title":"Fast-vqa: Efficient end-to-end video quality assessment with fragment sampling,","venue":null,"work_id":"b7382e6a-53ad-4d76-b767-f3cdf6f4b8c8","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.140757Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:cdeab2cd0ed53887e7ce84bd89d347a476495d5a9f7e928cb3f4fd6c8f1d2adf","observation_id":"a5d76651-a57b-4067-ab29-cab838afc2b4","resolution":{"observed_at":"2026-08-07T14:15:56.275262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.995545Z","title":"A deep learning based no-reference quality assessment model for ugc videos,","venue":null,"work_id":"86f9226f-7870-40af-b6f8-32cee121e802","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.223546Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:23019a960bff520dea27ccb3dfafa9e487430ada593e8f2788c42de9d76952f1","observation_id":"141736e6-6844-4070-9a96-8377261db06e","resolution":{"observed_at":"2026-08-07T14:15:56.115372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.857344Z","title":"Quality assessment of in-the-wild videos,","venue":null,"work_id":"9d5723e2-83b9-48c6-a927-c908b8e17626","year":2019},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.303727Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:e7b88ef5c687e8803272162d06f2fa47a4850e49ee775341f4f450f1e1ed06c7","observation_id":"075bcaea-ff09-4a68-b72e-52c2829fc70e","resolution":{"observed_at":"2026-08-07T14:15:55.923996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.720812Z","title":"Ugc-vqa: Benchmarking blind video quality assessment for user generated content,","venue":null,"work_id":"caf3c255-3f4d-41a5-9be7-57b4ccc5998f","year":2020},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.396067Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:81004300061577f2c4f69a8661a9a1142c13028236933fc9e0e0b448d1545769","observation_id":"db5585b8-a3c6-4ad3-8baa-48519105547e","resolution":{"observed_at":"2026-08-07T14:15:55.755136Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.610988Z","title":"Ve-bench: Subjective-aligned benchmark suite for text-driven video editing quality assessment,","venue":null,"work_id":"630887df-6b7a-4ed5-8147-193a140b71c3","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.493803Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:681b0b4822d855763f08723db6b513d0338f3b454cb16fc1369eb9efd223f511","observation_id":"015e49cc-725b-4faa-83a4-9c08c135c34f","resolution":{"observed_at":"2026-08-07T14:15:55.658082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.475777Z","title":"Subjective-aligned dataset and metric for text-to-video quality assessment,","venue":null,"work_id":"8e5252ed-49b6-4eaf-bf23-1d34196b4286","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.569424Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:eacb73aeff0db7d8b4701f0c58752d0a7839362c4abfd32ce96d40f765b31ceb","observation_id":"67dc21db-bbe0-4bd6-a807-dea04770e5b2","resolution":{"observed_at":"2026-08-07T14:15:55.566921Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.326401Z","title":"Aigv-assessor: Benchmarking and evaluating the perceptual quality of text-to-video generation with lmm,","venue":null,"work_id":"43af6dbb-f8d4-4c18-896f-3b79489a519d","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.660369Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:bdcfbf169ba37dbf1d26c8f508e961907e8626c6c0f3e078424a8854960e42bc","observation_id":"382ed968-9186-4e37-a265-2431fb21154b","resolution":{"observed_at":"2026-08-07T14:15:55.397791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.208835Z","title":"Cvpr 2023 text guided video editing competition,","venue":null,"work_id":"d17d9636-2d58-4dd5-b4c9-6de06ec6836c","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.715156Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:5e1cdc77f95c48320c490498d022a760b1807ebc76ec4229f4c4189bf9ec048a","observation_id":"37e63f66-13fb-4635-8466-3d8fe8f257ed","resolution":{"observed_at":"2026-08-07T14:15:55.263392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:55.063178Z","title":"Harnessing the power of llms in practice: A survey on chatgpt and beyond,","venue":null,"work_id":"881d4db7-c4a1-4366-a07d-858d622b98ef","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.772617Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:801fc26659e8b1daef77a23a956da4bf4433e7de80f2b3da0b72e41f49605c25","observation_id":"8567a7a2-5497-4f38-bc97-57584fa67d93","resolution":{"observed_at":"2026-08-07T14:15:55.127769Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.873579Z","title":"Señorita- 2m: A high-quality instruction-based dataset for general video editing by video specialists,","venue":null,"work_id":"3b5132b3-a237-404b-8baa-8f3c99dd343c","year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.864925Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:4e17f0569ac43bbec2b140b3173b1075d8b990a2eaced5442ceaedc4eccc8b69","observation_id":"909a4cb3-8658-48fb-9a50-7f876484b7a9","resolution":{"observed_at":"2026-08-07T14:15:54.957500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.766512Z","title":"The 2017 davis challenge on video object segmentation,","venue":null,"work_id":"837fb387-96e6-4e06-87c2-5a3abd6169e3","year":2017},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:45.952544Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:ea9ce999b13ddee0c1286eb762587d0c00ab69521595f4d50b1af79efa561684","observation_id":"fc602e93-00f8-4104-9789-58f37f95a2e1","resolution":{"observed_at":"2026-08-07T14:15:54.811565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.652967Z","title":"The kinetics human action video dataset,","venue":null,"work_id":"c32ab599-c39d-4637-a5f9-0147753e1643","year":2017},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.050651Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:9455a56e20d19b9a00790761cfe0cd9fe9ab9848882932b8456d3f9b467c1cee","observation_id":"b3a43984-edbb-4e33-8d52-60aa4f3a2ae7","resolution":{"observed_at":"2026-08-07T14:15:54.691934Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.501381Z","title":"JimengAI","venue":null,"work_id":"f5706de7-5780-4dc3-bb59-788eafa33701","year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.117830Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:5913b467235d64b61ae5da0e45f8702721e61b1875689f5fb0d0c2e24ae70776","observation_id":"a159a87b-60dd-4c58-8742-d0082cd328ed","resolution":{"observed_at":"2026-08-07T14:15:54.573138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.336329Z","title":"Methodology for the subjective assessment of the quality of television pictures,","venue":null,"work_id":"1c4e7919-f020-4d64-99a9-56f47dc81c94","year":2012},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.186786Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:066ef1108658475efbc6bee52046568e70f81f5a65b373ce842772c04517f508","observation_id":"e6e4e421-f9e6-476f-9d65-a954de5c28f3","resolution":{"observed_at":"2026-08-07T14:15:54.428466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.169105Z","title":"Blind image quality assessment based on high order statistics aggregation,","venue":null,"work_id":"a59ca9c3-6219-42b0-9a5a-770fc80eae89","year":2016},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.255534Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:3c9343d3eaf8574812dd6c834ad6d442b114b6a3cd528b5620e00d553bef96e8","observation_id":"c9cdea49-6a70-47ab-9afd-3a6c8b0baa73","resolution":{"observed_at":"2026-08-07T14:15:54.235971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.095797Z","title":"Learning without human scores for blind image quality assessment,","venue":null,"work_id":"511f3c93-bc3f-45ba-b1ec-3a71b00d67a0","year":2013},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.336487Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:90f2b257920be74d90d490553ab2be306d35ad07b761d4200dfdc9986bc42970","observation_id":"0fb5fcef-ca42-4cee-ad5b-f95dd1083574","resolution":{"observed_at":"2026-08-07T14:15:54.125353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:54.025445Z","title":"Making a “completely blind","venue":null,"work_id":"ebc60996-781d-427c-b8e8-b827ccc4657d","year":2013},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.415983Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:b21bc964ed894f5e09ebbcc9c4edcc985d9a3ad1c19747e73093d86d4d3b4d65","observation_id":"58668ad9-9f78-4c1b-b000-54e70d6d5511","resolution":{"observed_at":"2026-08-07T14:15:54.055030Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.983229Z","title":"Clipscore: A reference-free evaluation metric for image captioning,","venue":null,"work_id":"a96f5067-945e-4795-9b8c-cf7ec1aba230","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.508413Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:da7db342a10af84fe846c344909f4ac532491056f0beaba9394d6703841bc048","observation_id":"147b5037-7061-4135-a851-e193d80a0556","resolution":{"observed_at":"2026-08-07T14:15:53.999272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.01291","last_updated":"2024-06-18T07:09:55Z","snapshot_observed_at":"2026-07-06T17:54:07.815827Z","submitted_at":"2024-04-01T17:58:06Z","title":"Evaluating Text-to-Visual Generation with Image-to-Text Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.01291","snapshot_observed_at":"2026-08-07T14:15:46.589608Z","title":"Evaluating text-to-visual generation with image-to-text generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.589608Z"},"links":{"cited_paper":"/paper/2404.01291","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:6e30e730918ed0b6b84164107f466199af7f1a054d32ba9c6b753462efa82958","observation_id":"0082d370-eec2-4389-a745-f24588fd8b08","resolution":{"observed_at":"2026-08-07T14:15:46.589608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.08358","last_updated":"2025-04-11T08:46:49Z","snapshot_observed_at":"2026-08-07T16:06:37.023288Z","submitted_at":"2025-04-11T08:46:49Z","title":"LMM4LMM: Benchmarking and Evaluating Large-multimodal Image Generation with LMMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.08358","snapshot_observed_at":"2026-08-07T14:15:46.647524Z","title":"Lmm4lmm: Benchmarking and evaluating large-multimodal image generation with lmms,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.647524Z"},"links":{"cited_paper":"/paper/2504.08358","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:7ae524ddf0c2ae56872d23bc2afd0c0351d9ce5596affde898b27473ab358579","observation_id":"fa6aed1c-e178-4085-a73d-c7cc251bcf1c","resolution":{"observed_at":"2026-08-07T14:15:46.647524Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:46.707838Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.707838Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:881c609b62d1f916c08f7a8e7d14bc199ab2e84426bec17934d45df595cb1b2f","observation_id":"9720f9e0-4f87-4be1-ad52-edb8df975867","resolution":{"observed_at":"2026-08-07T14:15:46.707838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.877089Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale,","venue":null,"work_id":"a47b4189-53f8-4abd-9e23-56d2b755f39b","year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.796154Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:97880f2104dd8ff943d9dcba4425e95456d15cfde196b154ab54130404136524","observation_id":"b6ab865e-ded1-44f6-b9ce-1167a8e66290","resolution":{"observed_at":"2026-08-07T14:15:53.957123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.748793Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale,","venue":null,"work_id":"3b0db710-266e-4b83-9009-7f53a2f250d1","year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.889880Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:d1e7834d4feacf993c385e47c1eef65eef52f8ddef4ee7a11c6efb9b25d998fc","observation_id":"514cfcbf-2390-4329-9e73-c59ea492ffa2","resolution":{"observed_at":"2026-08-07T14:15:53.808845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.01601","last_updated":"2021-06-11T09:36:50Z","snapshot_observed_at":"2026-08-06T23:46:59.006674Z","submitted_at":"2021-05-04T16:17:21Z","title":"MLP-Mixer: An all-MLP Architecture for Vision","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.01601","snapshot_observed_at":"2026-08-07T14:15:46.972275Z","title":"Mlp-mixer: An all-mlp architecture for vision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:46.972275Z"},"links":{"cited_paper":"/paper/2105.01601","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:dddcf71753dcb3f205e4a37f133f9b93874eaadcedf192d40c4ada5646842f1c","observation_id":"2a2ac725-4e2f-4ca5-9db3-93e2c7a3cef6","resolution":{"observed_at":"2026-08-07T14:15:46.972275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.10270","last_updated":"2022-06-23T10:51:04Z","snapshot_observed_at":"2026-08-06T08:18:31.112514Z","submitted_at":"2021-06-18T17:58:20Z","title":"How to train your ViT? Data, Augmentation, and Regularization in Vision Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.10270","snapshot_observed_at":"2026-08-07T14:15:47.047930Z","title":"How to train your vit? data, augmentation, and regularization in vision transformers,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.047930Z"},"links":{"cited_paper":"/paper/2106.10270","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:6ad8a54c2d573a322a3afcdab2bbbe8f7e2235f40b549a5a754cb464cdade77f","observation_id":"9fc44e78-5762-44f8-b03f-022e60c60f95","resolution":{"observed_at":"2026-08-07T14:15:47.047930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.01548","last_updated":"2022-03-13T18:58:43Z","snapshot_observed_at":"2026-07-06T11:15:26.915483Z","submitted_at":"2021-06-03T02:08:03Z","title":"When Vision Transformers Outperform ResNets without Pre-training or Strong Data Augmentations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.01548","snapshot_observed_at":"2026-08-07T14:15:47.098749Z","title":"When vision transformers outperform resnets without pretraining or strong data augmentations,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.098749Z"},"links":{"cited_paper":"/paper/2106.01548","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:4719baa3df409156651f20f13c772b19915d62156cec6b62110713e079b9ce36","observation_id":"8febaff3-2ab2-4aa5-b1d0-617cdeb1442c","resolution":{"observed_at":"2026-08-07T14:15:47.098749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.500783Z","title":"Surrogate gap minimization improves sharpness-aware training,","venue":null,"work_id":"216e36e9-7542-4ffd-afd3-f87aa1dcbad7","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.146349Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:44dffce774c875f190d58bf45a6b151d98f47745e79c1a3dfa0d331386f68515","observation_id":"0db7e7a8-44cb-495a-88e1-b7ee9eb8c641","resolution":{"observed_at":"2026-08-07T14:15:53.597132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.341023Z","title":"Lit: Zero-shot transfer with locked-image text tuning,","venue":null,"work_id":"62cd42ed-3e31-4080-8103-5cae877e2b67","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.196903Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:ec31bc26c1f5e854ca38a2d24a7a10a12de542e2522f416169b79a18b51ef788","observation_id":"761438d2-410b-48c9-a6b6-59cb3f6a55f7","resolution":{"observed_at":"2026-08-07T14:15:53.438080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-07T14:15:47.249382Z","title":"Qwen-vl: A versa- tile vision-language model for understanding, localization, text reading, and beyond,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.249382Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:39a66d0ce89d40d7af1b44de9cf05344b8c1fccec976a74c056fe7d8592cc5d2","observation_id":"1664f839-5ebe-4a7c-bb8f-b579c62ce79f","resolution":{"observed_at":"2026-08-07T14:15:47.249382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:53.125365Z","title":"Mlp-net: Multilayer perceptron fusion network for infrared small target detection,","venue":null,"work_id":"b479b6b3-fba3-4ea4-b019-93e78834e81f","year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.303914Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:c5f16c24ec53dfeb82ed82a65eb76133179b85f3efcdcde92f7a23d679630db7","observation_id":"8adb38d0-4237-4e12-b5bd-e0eb6b3fe445","resolution":{"observed_at":"2026-08-07T14:15:53.213820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:52.877682Z","title":"Lora: Low-rank adaptation of large language models,","venue":null,"work_id":"400d4d09-f4f4-43c2-bd09-2ee5f7acb93a","year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.364459Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:1bf57327b20005b5018c5a7c3821069a671306b2a5b02d2af472a0e69beed0f5","observation_id":"65a60f14-dd82-401e-a048-46d7db18602a","resolution":{"observed_at":"2026-08-07T14:15:52.987257Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:52.667618Z","title":"Imagereward: learning and evaluating human preferences for text-to-image generation,","venue":null,"work_id":"fc2eaa4f-8117-4358-96a0-cc338a019a4d","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.419599Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:c63f32e57459ad18e0c64af8758b0bc51d28afde003bd930994fdc35ed716738","observation_id":"d53a406b-4189-488a-99ed-9272ada4c374","resolution":{"observed_at":"2026-08-07T14:15:52.781780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:52.498420Z","title":"Blip: Bootstrapping language-image pre-training for unified vision- language understanding and generation,","venue":null,"work_id":"648bb444-4dbe-437d-957a-6bed02372e90","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.497508Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:e361b5f75b1d643dad760f5f8b8a02b5ed08895dbf96bd0d03d2f8a32459bf4e","observation_id":"c95245d3-7d6c-402a-aa39-aec719e0a493","resolution":{"observed_at":"2026-08-07T14:15:52.561473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:52.275626Z","title":"Pick-a-pic: An open dataset of user preferences for text-to-image generation,","venue":null,"work_id":"793a3e92-43f8-45fb-bd11-465c62940c8b","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.556046Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:ffcd40f085bf8784fda57f0cf90a5ca240108160cdc91f46a4f173e734b3d67b","observation_id":"4f49277c-9c6d-4030-a140-a3ee6b1f4dde","resolution":{"observed_at":"2026-08-07T14:15:52.397957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:52.078567Z","title":"Building cnn-based models for image aesthetic score prediction using an ensemble,","venue":null,"work_id":"8602fdd4-ddd7-400c-a98a-c8bf11d5679f","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.618742Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:f0088130c963cc5153e1a85f0b586a308b9cbc2cfa16cd0d38a453c8c24a9c3f","observation_id":"11517637-b9b8-4fb5-a164-d3be77ecd053","resolution":{"observed_at":"2026-08-07T14:15:52.174546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:51.827935Z","title":"Llava-next: A strong zero-shot video understanding model,","venue":null,"work_id":"a15bfb9d-fd24-4cef-9f26-bb4e30feecf9","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.671411Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:9d63961b0d07a1cc28053886448a58096ba4b7deeeeac212d97eee3d7f1eccaa","observation_id":"f8b2128c-3ff8-4a5c-b537-99a95b3bab61","resolution":{"observed_at":"2026-08-07T14:15:51.951951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:51.615212Z","title":"Internvideo: General video foundation models via generative and discriminative learning,","venue":null,"work_id":"3f99eea7-0861-4aca-ae0b-0b37619ca54a","year":2022},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.723918Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:91947325611d45eae5c45a7790fd529c3db68c0f92b36b4fd5d9edef1e0ad11b","observation_id":"92317989-0820-4ead-b432-384fdfa782e0","resolution":{"observed_at":"2026-08-07T14:15:51.692478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13106","snapshot_observed_at":"2026-08-07T14:15:47.793673Z","title":"Videollama 3: Frontier multimodal foundation models for image and video understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.793673Z"},"links":{"cited_paper":"/paper/2501.13106","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:5dcc04c8fdd5ab7858b9118c5fced350e6eb2ebe77ac8e6713a52bdb3c6f3c4c","observation_id":"b97954b8-5d3e-4053-a390-0a9036c01158","resolution":{"observed_at":"2026-08-07T14:15:47.793673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:51.406576Z","title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks,","venue":null,"work_id":"5a1b81ee-3750-4dbf-bd8d-5287c1d8b4ec","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.842241Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:fb5fe99753d76d339d35198ee9e06ff79bce7f21ae01a8dd913e67b10a458d01","observation_id":"16b6cae0-883d-4759-b36e-d73e8eec2a68","resolution":{"observed_at":"2026-08-07T14:15:51.511413Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:51.273234Z","title":"mplug-owl3: Towards long image-sequence understanding in multi-modal large language models,","venue":null,"work_id":"79e9afc2-d23f-4768-ad7e-565fc9f86238","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.905970Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:c370761a20efd3a26cd6ea1e93439b1eaada4ad6474f374df0112dcb7d1a4b39","observation_id":"dbc1d83f-1c7a-43b1-995f-c66bac60f979","resolution":{"observed_at":"2026-08-07T14:15:51.344286Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:47.959442Z","title":"No-reference image quality assessment in the spatial domain,","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:47.959442Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:1cbfc122be6e0a4d0dc7b1acd6ab54a154ce4e31048d33f37220af2470106428","observation_id":"caaec614-1d36-46d3-b353-25615a657f3e","resolution":{"observed_at":"2026-08-07T14:15:47.959442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:51.115921Z","title":"Blind image quality estimation via distortion aggravation,","venue":null,"work_id":"84b941e4-0776-4dae-b623-c918f975d868","year":2018},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.029727Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:0c952eaa1d32cb3ef54018a58c357149144d5900f3236fc6197ab38520daeb7b","observation_id":"e78e9d04-ab79-454a-9fff-8fdf3e641069","resolution":{"observed_at":"2026-08-07T14:15:51.198918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:50.910319Z","title":"Blind quality assessment based on pseudo- reference image,","venue":null,"work_id":"048bc7c2-c120-4c44-a0c7-c7611f900958","year":2018},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.090491Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:41c5a97632536abb3dc036ee4372416e8f891b2cf8700bca66999510431d676c","observation_id":"79cbeca4-a14e-441e-be03-e04a34a93f5b","resolution":{"observed_at":"2026-08-07T14:15:51.004024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:48.138760Z","title":"Visual instruction tuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.138760Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:6f0f8d9503163e7c556901c133923305bf14f5ce75da86fb54778fe3963427d8","observation_id":"403b8b0a-6ae7-426a-a442-ceabac5f51bc","resolution":{"observed_at":"2026-08-07T14:15:48.138760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07895","last_updated":"2024-07-28T19:58:08Z","snapshot_observed_at":"2026-07-06T18:44:24.873040Z","submitted_at":"2024-07-10T17:59:43Z","title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.07895","snapshot_observed_at":"2026-08-07T14:15:48.190049Z","title":"Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.190049Z"},"links":{"cited_paper":"/paper/2407.07895","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:c5b80895a01bc59b74360834b90a1584967eecfe7447a035257d0baff9123462","observation_id":"1f8005a3-abc0-497d-a663-8d9b958afd0b","resolution":{"observed_at":"2026-08-07T14:15:48.190049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:48.245414Z","title":"Llava-next: Improved reasoning, ocr, and world knowledge,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.245414Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:e3dadef2b8a2a225d60f552991ede961a0fa60b8b2b63b1a7b8bb6f99c9f9f0b","observation_id":"f883cf17-f2ed-408c-9b71-becbad8dd081","resolution":{"observed_at":"2026-08-07T14:15:48.245414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:48.349429Z","title":"Improved baselines with visual instruction tuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.349429Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:6e53f1a7e94e2ff7bdd340603f7531999493f2fb1689ea8b943dc0381fd5442c","observation_id":"a9df28ab-111c-4a68-821e-a7aaf4172c19","resolution":{"observed_at":"2026-08-07T14:15:48.349429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:50.657726Z","title":"Llava-next: Stronger llms supercharge multimodal capabilities in the wild,","venue":null,"work_id":"7551d759-9346-40ae-a132-1ba091cc6efc","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.412809Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:e62c9d0743da15dfbdbd7bb5d3671b62cc9138205c739a87a8ba2e7b2ecfe42d","observation_id":"865709a4-8de1-4b48-be07-011908404bcd","resolution":{"observed_at":"2026-08-07T14:15:50.758104Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:50.430836Z","title":"Llava-next: What else influences visual instruction tuning beyond data?,","venue":null,"work_id":"d7300167-9967-4430-981c-bccc320c9abe","year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.473759Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:f60da8c569f658e5b6689ec9812d721570867041d51f1397d970726b94966226","observation_id":"8b7fa63b-bfa7-40b9-993b-d750b86428ca","resolution":{"observed_at":"2026-08-07T14:15:50.534207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-07T14:15:48.537796Z","title":"Video-llama: An instruction-tuned audio-visual language model for video understanding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.537796Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:7ff90aefb6fa2c1452dea01369ef3e3aaf63d9554ba057efef20c85aea85876d","observation_id":"18370840-77bd-42b7-9442-253997febc9a","resolution":{"observed_at":"2026-08-07T14:15:48.537796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-07T14:15:48.595108Z","title":"Videollama 2: Advancing spatial-temporal modeling and audio understanding in video-llms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.595108Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:74262ff6257c576ddb933c5e20abba36ce09227353e2cf9e36cb877cf99a805f","observation_id":"2a484005-937b-47b3-912e-9e815524a744","resolution":{"observed_at":"2026-08-07T14:15:48.595108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:50.237089Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":"8ca92e43-7b6e-4736-9670-4a187b907905","year":2021},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.673221Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:6c6bbced5c5f006936c05a1d0bce93ed609a6b37b496c00e8474d759a4fda2f1","observation_id":"3e219ea3-4a27-42f1-9288-8bbe7e7f46c3","resolution":{"observed_at":"2026-08-07T14:15:50.332408Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:49.999171Z","title":"Unmasked teacher: Towards training- efficient video foundation models,","venue":null,"work_id":"b98310b4-809b-4388-acc6-9c02f8583113","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.732449Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:32111377377a11641b02bf7dc39a2ebdae2f828453b266b20e39ce5578bbed39","observation_id":"dba5ab64-a687-4600-bbb6-142a5c5a19b2","resolution":{"observed_at":"2026-08-07T14:15:50.106302Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:49.782748Z","title":"Stablevqa: A deep no-reference quality assessment model for video stability,","venue":null,"work_id":"30a4e065-0867-4c8c-94e7-2e9403e74d06","year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.810810Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:425433b4758cb2c739c87967590519bba309b83e6ec407c2b4f4cc6c366d671e","observation_id":"7e7a514f-68bf-41ce-aea6-7b24568ecc26","resolution":{"observed_at":"2026-08-07T14:15:49.870172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17090","last_updated":"2023-12-28T16:10:25Z","snapshot_observed_at":"2026-08-02T07:14:02.308302Z","submitted_at":"2023-12-28T16:10:25Z","title":"Q-Align: Teaching LMMs for Visual Scoring via Discrete Text-Defined Levels","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17090","snapshot_observed_at":"2026-08-07T14:15:48.881635Z","title":"Q-align: Teaching lmms for visual scoring via discrete text-defined levels,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.881635Z"},"links":{"cited_paper":"/paper/2312.17090","citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:aacc45072ee2668fd62642340de5303a257c806312407de6b4f246b6aafe36f6","observation_id":"f8d41227-00e3-4ca1-8fdf-781525ee55a7","resolution":{"observed_at":"2026-08-07T14:15:48.881635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:49.541531Z","title":"Excellent","venue":null,"work_id":"b4414455-bba7-4130-b398-891fa4564aca","year":null},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.932160Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:af3b72dac2ea5d9fe868ad158b78fef273b24c9f5e4298b4a7a8f26af6f9d96b","observation_id":"b589e717-2e61-441e-bf33-a1400fd02fa0","resolution":{"observed_at":"2026-08-07T14:15:49.655959Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:15:49.309914Z","title":"Excellent","venue":null,"work_id":"a2ef180a-6a07-480f-98c7-8cef90c1a51a","year":2016},"citing_paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:48.995779Z"},"links":{"citing_paper":"/paper/2505.19535"},"observation_digest":"sha256:04423f66ce579069144bb0690c6d33194f06d951f0d804e143bedd2236af0a13","observation_id":"4d17d73d-ae3a-467f-bceb-a0c55182ad0f","resolution":{"observed_at":"2026-08-07T14:15:49.409764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.19535","last_updated":"2025-05-26T05:47:09Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T22:51:52.398041Z","submitted_at":"2025-05-26T05:47:09Z","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs"},"reference_resolution":{"displayed":71,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":55},"total_outbound_references":71},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 71 of 71 outbound references and 0 inbound Pith citation observations for arXiv:2505.19535."}