{"as_of":"2026-08-13T15:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b8718b2dd13e8dcf742b31669306a718c4cc788637b1730718897cdf0b383360","coverage":[{"denominator":62,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":62,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:39:34.111978Z","state":"measured"},{"denominator":63,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":63,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:39:28.786395Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T15:39:34.749620Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"cited_work":{"arxiv_id":"2505.14336","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.14336","snapshot_observed_at":"2026-08-07T15:39:34.749620Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","venue":"eess.AS","work_id":"ddd8a168-6998-4230-a53b-dd483c9961b9","year":2025},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:28.786395Z"},"links":{"cited_paper":"/paper/2505.14336","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:828f2488a76e9b065512846c1e274d2bed18905c9c8daf82662e53ccced183c9","observation_id":"eb34c564-0255-47f0-a5af-47e66d1a49da","resolution":{"observed_at":"2026-08-07T15:39:34.821428Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.14336/citation-record","integrity":"/paper/2505.14336/integrity","json":"/paper/2505.14336/citation-record.json","paper":"/paper/2505.14336"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"cited_work":{"arxiv_id":"2505.14336","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.14336","snapshot_observed_at":"2026-08-07T15:39:34.749620Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","venue":"eess.AS","work_id":"ddd8a168-6998-4230-a53b-dd483c9961b9","year":2025},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:28.786395Z"},"links":{"cited_paper":"/paper/2505.14336","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:828f2488a76e9b065512846c1e274d2bed18905c9c8daf82662e53ccced183c9","observation_id":"eb34c564-0255-47f0-a5af-47e66d1a49da","resolution":{"observed_at":"2026-08-07T15:39:34.821428Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:43.695743Z","title":"This is crucial in resource-constrained LLM-based A VSR systems, as we aim to improve performance despite using smaller-scale LLMs and pre-trained encoders","venue":null,"work_id":"06ca1b82-555e-498f-a0db-f17b2eb2b0f9","year":null},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:28.842189Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:b50cbec1ed9117596cd42107937f2e26c1c5f338f1dcc10b1a890d6417355878","observation_id":"854669e9-b19b-4a3d-afb6-81d7d591a698","resolution":{"observed_at":"2026-08-07T15:39:43.754098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:43.578374Z","title":"Transcribe{task prompt}to text","venue":null,"work_id":"3cb67e30-9418-4604-b430-6d39d1ea7ea6","year":null},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:28.959989Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:7d6a03e25f8eb90fe5d450079ab0d2e093d422df61d89e850166decd5a169ee1","observation_id":"b7ab5303-54d5-4ca6-9a85-233cd69bf06c","resolution":{"observed_at":"2026-08-07T15:39:43.622848Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:43.429210Z","title":"Its key innovation is replacing the lin- ear projector with a Top-K sparse MoE module","venue":null,"work_id":"87857eba-2f15-486e-a1b7-f886c5614663","year":2020},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:29.067190Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:3ec9634b86896436c6f2b754eb8db5a6bb9f8a37747b7a80cab8e1c998fb0d6b","observation_id":"4c0562e3-906b-4377-854b-4debf0c19cc5","resolution":{"observed_at":"2026-08-07T15:39:43.513781Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.01037","last_updated":"2023-09-25T01:20:23Z","snapshot_observed_at":"2026-08-13T12:33:46.085209Z","submitted_at":"2023-03-02T07:47:18Z","title":"Google USM: Scaling Automatic Speech Recognition Beyond 100 Languages","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.01037","snapshot_observed_at":"2026-08-07T15:39:29.177686Z","title":"Google usm: Scaling automatic speech recog- nition beyond 100 languages,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:29.177686Z"},"links":{"cited_paper":"/paper/2303.01037","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:a39fb50eddeb24ef49497e93e98dfbc85534dcc3a7b9f260798435c13435e233","observation_id":"48b325f5-22a0-4758-bcab-bcf4935c0bb3","resolution":{"observed_at":"2026-08-07T15:39:29.177686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:43.178233Z","title":"Deep speech 2: End-to-end speech recognition in english and mandarin,","venue":null,"work_id":"69028654-112d-4f6b-b51c-5a9d89379923","year":2016},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:29.296827Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:8f7c83b70854be0ecee516460ee8f325796d0fb493f21720443bfab491bf2f25","observation_id":"6af2e464-b6fa-45be-8ebc-79f4ad2c0280","resolution":{"observed_at":"2026-08-07T15:39:43.274160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:42.974470Z","title":"End-to-end speech recognition: A sur- vey,","venue":null,"work_id":"94257293-1004-4318-a52d-9dbf60bc785f","year":2023},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:29.392445Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:be9926e86898b7cf0af19f0814dc53fd628543f9bc3a7bf4910f601800979689","observation_id":"77e79aba-3861-4ae5-a3ff-d9b6241e220a","resolution":{"observed_at":"2026-08-07T15:39:43.064274Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:42.861628Z","title":"Audio-visual speech modeling for continuous speech recognition,","venue":null,"work_id":"8e3a2182-fab6-443b-845e-1ab2175a92d2","year":2000},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:29.512583Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:fb8d2b8c62636b8a49ddb45fd0ed11b01088203084c6827abc3f2b9230fcce16","observation_id":"a7ebaeff-76fa-40d6-beee-51a715d9f196","resolution":{"observed_at":"2026-08-07T15:39:42.921045Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:42.741556Z","title":"Investigation of speech separation as a front- end for noise robust speech recognition,","venue":null,"work_id":"96760db4-9534-46c5-8fc4-afbdcca9031c","year":2014},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:29.617090Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:f714d38d9f2e73d6bcc70b77389a0144477a38def340c30e301d3a52ec3014bc","observation_id":"188010e7-f95a-4c3e-a6b5-424bef2cbb2e","resolution":{"observed_at":"2026-08-07T15:39:42.815885Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:42.542159Z","title":"Audio-visual speech recognition using deep learn- ing,","venue":null,"work_id":"8900ee52-99b8-4814-b859-60e2b8bcc107","year":2015},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:29.718054Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:195c96354abbef1cce10febf83e34b7dd3d81c8b4b8e60e83590e7f4cf012f47","observation_id":"4af7a8d2-2900-4128-8e72-eb6ca82d3e0d","resolution":{"observed_at":"2026-08-07T15:39:42.636962Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:42.411916Z","title":"Deep audio-visual speech recognition,","venue":null,"work_id":"05c31d2a-041a-48c3-bddc-f3cc563b055c","year":2018},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:29.793596Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:e0e8147202679c9a5e4a7f4f996454c28c806d9c72d6bbf9736272e4dca91bdb","observation_id":"35050d51-0909-49ee-bb22-e1d9811c4f07","resolution":{"observed_at":"2026-08-07T15:39:42.460290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:42.250806Z","title":"Audio-visual speech recognition with a hybrid ctc/attention architecture,","venue":null,"work_id":"6f627adc-8131-49b1-a678-e57db863aab3","year":2018},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:29.884046Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:1b211b72a13c14dd95a1939444f128bd37996b77636b9549837751cff8a35cf7","observation_id":"4162f35f-1e8f-4e3a-a19e-5cce765eb734","resolution":{"observed_at":"2026-08-07T15:39:42.312711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:42.016109Z","title":"End-to-end audio-visual speech recognition with conformers,","venue":null,"work_id":"ab9aae97-b1be-410d-990d-e4e53f6156eb","year":2021},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:29.960022Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:d95abbed02940fdff90621657f1d694f822bd9ece36749734dd533920e1476ff","observation_id":"cb370472-9c29-485b-a705-d25f88a2af39","resolution":{"observed_at":"2026-08-07T15:39:42.134659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:41.718457Z","title":"Visual context-driven audio feature enhancement for robust end-to-end audio-visual speech recognition,","venue":null,"work_id":"9bfa3d71-8f96-428a-8946-4de95fe9d928","year":2022},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.022183Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:96eb9de11633b48ad0b0fdf90e8b33ad6f0750bcb0d01e7866eaf4c01c385bd4","observation_id":"ebdfcbc3-9b9c-4648-bee3-ab9970c64991","resolution":{"observed_at":"2026-08-07T15:39:41.826894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:41.409807Z","title":"Auto-avsr: Audio-visual speech recognition with automatic labels,","venue":null,"work_id":"dfefdda8-bd7a-47d9-b989-f7d1423addea","year":2023},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.099358Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:3b5f7040355be055ccd9984c6ad03b05bd394b8d3daee38d648c683a5f3da6a8","observation_id":"9f0b6873-112f-4b70-9cf0-599666223be1","resolution":{"observed_at":"2026-08-07T15:39:41.559836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:41.163996Z","title":"Whisper-flamingo: Integrating visual features into whisper for audio-visual speech recognition and translation,","venue":null,"work_id":"0c0605d2-8f93-481b-a930-c0e12fba5692","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.179611Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:3532d56db563913425b3da897b03fadb9615849d40749b0611abbd361a823e0e","observation_id":"96890755-e6b6-4be8-b254-2881b9b49b1b","resolution":{"observed_at":"2026-08-07T15:39:41.292636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:40.831921Z","title":"A survey on self-supervised learning: Algorithms, applications, and future trends,","venue":null,"work_id":"098ffb06-1e51-4dd0-bef5-9787ea9b0e9e","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.265926Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:e5165aa79e9669407a7af046518b0552c23938bd72627fab39e5496ec74457b6","observation_id":"d83c6aa0-e7a5-492c-a3e1-f8c869c0f0b9","resolution":{"observed_at":"2026-08-07T15:39:40.975092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:40.629100Z","title":"Learning audio-visual speech representation by masked multimodal cluster prediction,","venue":null,"work_id":"5bd2ada7-5d05-4218-b8f8-8f232fccd203","year":2022},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.329908Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:d1e1d6d38dfd740b50f6b0aba297b0206ae542631af4985296da1df90771edf6","observation_id":"4242120d-d12d-4b0a-b055-bdd8319be364","resolution":{"observed_at":"2026-08-07T15:39:40.725609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:40.401404Z","title":"Jointly learning visual and auditory speech representations from raw data,","venue":null,"work_id":"b11b4f16-f5fc-4404-8f62-57beabff81ef","year":2023},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.413872Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:75bf69b9cbe347efccb902f41c7ef9f61f8f9f3031c5384019263302a64d507d","observation_id":"c333ce47-1df5-4dbe-9ef5-2ee24958b68e","resolution":{"observed_at":"2026-08-07T15:39:40.495493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:40.209936Z","title":"u-hubert: Unified mixed-modal speech pretraining and zero-shot transfer to unlabeled modality,","venue":null,"work_id":"2695986a-c86a-4247-ab64-d20731c9cc8f","year":2022},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.498623Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:f0a65e186c890916315801186c6872c5d60e2603d6dc126e9420acce234567d7","observation_id":"5c8e2a37-9a4f-4736-a29b-c404690f6afa","resolution":{"observed_at":"2026-08-07T15:39:40.289862Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:40.013832Z","title":"Braven: Improving self-supervised pre- training for visual and auditory speech recognition,","venue":null,"work_id":"c3582366-68ea-45b9-9281-77029bf2df8d","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.583391Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:028ef7f406aaab17511522d3ac2e0fdf804acc18ec7a284569835ee89dcf7d47","observation_id":"2bbe6460-a9ef-466d-a12c-3fded2e3cf02","resolution":{"observed_at":"2026-08-07T15:39:40.122153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:39.857035Z","title":"Unified speech recognition: A single model for auditory, visual, and audiovisual inputs,","venue":null,"work_id":"77f931a3-039d-4aa0-a696-5402add696f4","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.647934Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:7eec13ca7b54face6feee655e019bc8c1e2c01f51f5317a16d2d0650a7d616d2","observation_id":"a089d9ec-d62e-4726-ada1-d02a18e3969d","resolution":{"observed_at":"2026-08-07T15:39:39.932682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T15:39:30.744252Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.744252Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:82bc4c67ee119812b68ac2007fcc218114affdd7e33d032eefb94c44198304d5","observation_id":"feaadf66-b34a-4cb0-8f27-6aa4e0bab2d8","resolution":{"observed_at":"2026-08-07T15:39:30.744252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-07T15:39:30.816401Z","title":"Llama: Open and efficient foundation lan- guage models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.816401Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:70f6fe0f378f076c13b60c13bcf4169663eb97fb18003d9a07883ba1ce0a8778","observation_id":"fec5a96e-b54c-437f-b06c-851942001bb4","resolution":{"observed_at":"2026-08-07T15:39:30.816401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:39.695906Z","title":"Improved baselines with visual instruction tuning,","venue":null,"work_id":"edbab09f-2434-4ffa-b49c-53337b4f3841","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.899234Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:06a4af3f8e3d4a5514d55e4c327b026f5b192ebd8c58bb53994ee76bf1fd6600","observation_id":"9f279aed-e8f4-4d24-80a7-c207d0f18020","resolution":{"observed_at":"2026-08-07T15:39:39.770707Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:30.987118Z","title":"On generative spoken language modeling from raw audio,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:30.987118Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:dbd797a9f224b2bda138304d61575049ac685ab5e5e55a56cb79a114fe7df2cc","observation_id":"d470f4f4-f794-4237-a9a3-c508819393ac","resolution":{"observed_at":"2026-08-07T15:39:30.987118Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:39.517789Z","title":"Audiogpt: Understanding and generating speech, music, sound, and talking head,","venue":null,"work_id":"9f2e0819-3ecf-4e63-8287-e1d59e750243","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.081916Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:265671dec70e01dc080569ff69fb160d707ee037dad780a44f27da9dc62a82eb","observation_id":"72bd5ed8-d435-45e2-8f0d-ac21070670db","resolution":{"observed_at":"2026-08-07T15:39:39.596788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:39.345963Z","title":"Let’s go real talk: Spoken dialogue model for face- to-face conversation,","venue":null,"work_id":"128326ba-7676-40da-bef8-0a43e645bbbf","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.143555Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:085906752576515582524ea81dbcf79956627a380c3d586d8dc862f7a403e013","observation_id":"408ee907-d152-4960-835c-8966203d05aa","resolution":{"observed_at":"2026-08-07T15:39:39.428741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:39.114063Z","title":"Developing instruction-following speech language model without speech instruction-tuning data,","venue":null,"work_id":"23984f0d-89c4-4f91-9e56-0ef6bc58c62b","year":2025},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.222800Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:67dd1ff327ff0282e9f00029b8564e3e39054bf6f3478ee4e90175c3d9a55215","observation_id":"8170e4d9-a3c6-43a0-8b53-310b3f5797dd","resolution":{"observed_at":"2026-08-07T15:39:39.229595Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00168","last_updated":"2025-05-17T15:25:50Z","snapshot_observed_at":"2026-08-13T03:45:43.978778Z","submitted_at":"2024-09-30T19:17:46Z","title":"SSR: Alignment-Aware Modality Connector for Speech Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00168","snapshot_observed_at":"2026-08-07T15:39:31.292721Z","title":"Ssr: Alignment-aware modality connector for speech language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.292721Z"},"links":{"cited_paper":"/paper/2410.00168","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:2f693f192a0970c2d70c49ef3c114eb74ce6388a0900f8257bb57fc4c9264bc6","observation_id":"b49faad6-117c-4d7a-94ac-ada9c3300905","resolution":{"observed_at":"2026-08-07T15:39:31.292721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:38.936677Z","title":"It’s never too late: Fusing acoustic information into large language models for automatic speech recognition,","venue":null,"work_id":"cb43e6fe-e352-4d4f-be7a-c35b22669a56","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.368874Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:59a0cf0f6e093800e1d4444eccd1ea4aa3794f22dfe012eda12943574c92b977","observation_id":"7835cb45-c654-45f2-adb9-91e41bf36593","resolution":{"observed_at":"2026-08-07T15:39:39.011320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:38.677149Z","title":"Large language models are efficient learners of noise-robust speech recognition,","venue":null,"work_id":"d29fdcf1-605a-4d95-9920-6b5ad199892c","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.451865Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:525a45f2c2569a4804355379592d9ba10d1000454a041603f0b37c88b8db715b","observation_id":"3043acb5-fa44-4907-9105-98c275627aef","resolution":{"observed_at":"2026-08-07T15:39:38.784593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08846","last_updated":"2024-02-13T23:25:04Z","snapshot_observed_at":"2026-08-13T04:19:27.486516Z","submitted_at":"2024-02-13T23:25:04Z","title":"An Embarrassingly Simple Approach for LLM with Strong ASR Capacity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08846","snapshot_observed_at":"2026-08-07T15:39:31.525701Z","title":"An embarrassingly simple approach for llm with strong asr capacity,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.525701Z"},"links":{"cited_paper":"/paper/2402.08846","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:d762b5162a2fbb518edf6ff8d0ce9a533f3eeaae52cc8275e35c5953c3da2e84","observation_id":"fde77f27-3377-4510-a71b-9d197e0e108b","resolution":{"observed_at":"2026-08-07T15:39:31.525701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:38.482828Z","title":"Connecting speech encoder and large language model for asr,","venue":null,"work_id":"1189d42a-be8d-491f-b4af-af12e6a576da","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.586155Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:06271e27951caf44631568c9d2c5db7c3bb24944b28a7a9d72486e63e73aa835","observation_id":"9f32b6a4-acba-4693-adaf-d1b547c9d30f","resolution":{"observed_at":"2026-08-07T15:39:38.586326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:38.300625Z","title":"Prompting large language models with speech recognition abilities,","venue":null,"work_id":"d00ed2da-46d5-462d-8afc-9e700dd13220","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.667864Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:359f605b2cfc19ff62f9b807d628e93d4f605c9c5bc15810cd7ad3d161e13459","observation_id":"b27c3acf-289d-4f57-b72d-08758ab950f9","resolution":{"observed_at":"2026-08-07T15:39:38.393062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:38.086790Z","title":"Where visual speech meets language: Vsp-llm framework for efficient and context-aware visual speech process- ing,","venue":null,"work_id":"6347a249-d85c-415d-b2f0-eb8c57076906","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.772230Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:9038f2b136a7dab320e0a3d88e92d7b4d640cdeb3b2463a428774961e21c6dd6","observation_id":"e8b5be17-de52-4402-9183-5d8945b7a122","resolution":{"observed_at":"2026-08-07T15:39:38.209014Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:37.881366Z","title":"Large language models are strong audio- visual speech recognition learners,","venue":null,"work_id":"9b1d3af6-f1ba-40b2-8d51-cff632fad43a","year":2025},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.860619Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:9edec4257e47d8998e831f109233a1bc3eccad89e04c2e1e2f9117d1bd232067","observation_id":"4bd81aca-8e3e-4140-8c0c-b766d7a78aea","resolution":{"observed_at":"2026-08-07T15:39:37.984074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06362","last_updated":"2025-08-06T17:41:48Z","snapshot_observed_at":"2026-08-07T17:19:43.747391Z","submitted_at":"2025-03-09T00:02:10Z","title":"Adaptive Audio-Visual Speech Recognition via Matryoshka-Based Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2503.06362","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.06362","snapshot_observed_at":"2026-08-07T15:39:34.525624Z","title":"Adaptive Audio-Visual Speech Recognition via Matryoshka-Based Multimodal LLMs","venue":"cs.CV","work_id":"a1b20bb6-09da-4fc1-b83a-b7d134ed3c2b","year":2025},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:31.946915Z"},"links":{"cited_paper":"/paper/2503.06362","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:2ac11fef2f3c20e8cbb7e2759243732bfc3519cb5b950cb5a0efd13519eed9d9","observation_id":"74c03ad7-df57-4179-97a5-233f70fed895","resolution":{"observed_at":"2026-08-07T15:39:34.600776Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:37.662632Z","title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer,","venue":null,"work_id":"096b44a8-7216-41d3-a553-58293798f8c4","year":2016},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:32.029232Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:d2f1bd208331d16be92fc560b8703c40f2c4cdd23cf00a4c1b1dbb67dedd7dbb","observation_id":"0fded2c8-7c41-4daf-8b4e-16ea2954e66a","resolution":{"observed_at":"2026-08-07T15:39:37.759019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:37.468242Z","title":"Gshard: Scaling giant models with condi- tional computation and automatic sharding,","venue":null,"work_id":"63f9976b-ac76-4add-b2dd-b93d6e34d6d7","year":2021},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:32.102664Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:b1306e5593cf755f4990f5376c66fe2f9dbae5263e4016371352446c036ab547","observation_id":"f6b4946b-c23a-41ea-82be-6456a319d174","resolution":{"observed_at":"2026-08-07T15:39:37.572919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:37.253089Z","title":"Efficient fine-tuning of audio spectrogram transformers via soft mixture of adapters,","venue":null,"work_id":"571eafe2-0903-42c3-bcf7-fe8842f5bafc","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:32.197118Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:8715a442d58d58bcac512b60b580ea42ac5593641f8202542077bc16cd72f5fc","observation_id":"ee17e29d-dfee-4a80-9d27-a387a5a7a269","resolution":{"observed_at":"2026-08-07T15:39:37.370592Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04434","last_updated":"2024-06-19T06:04:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-07T15:56:43Z","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04434","snapshot_observed_at":"2026-08-07T15:39:32.285143Z","title":"Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:32.285143Z"},"links":{"cited_paper":"/paper/2405.04434","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:b54c0e73806f3265e54fcce8b95747148104d03bc42714aab12164c1d2a0f349","observation_id":"ba451649-cbec-430b-9c6f-ea5250ed1c15","resolution":{"observed_at":"2026-08-07T15:39:32.285143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04153","last_updated":"2024-07-04T20:59:20Z","snapshot_observed_at":"2026-08-12T23:28:33.412949Z","submitted_at":"2024-07-04T20:59:20Z","title":"Mixture of A Million Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04153","snapshot_observed_at":"2026-08-07T15:39:32.373581Z","title":"Mixture of a million experts,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:32.373581Z"},"links":{"cited_paper":"/paper/2407.04153","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:da8d6f321866118c9a08a04fbf8b36eb784fb138d545b2320c9837ac4cb0a4d9","observation_id":"c6287b0f-f201-490d-b5b4-5e8dc12f5861","resolution":{"observed_at":"2026-08-07T15:39:32.373581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:37.095785Z","title":"Olmoe: Open mixture-of-experts lan- guage models,","venue":null,"work_id":"da36737c-8406-48e4-8db2-c0f50d03df04","year":2025},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:32.488903Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:131a7f3a6263d804d17c9432b23025d0e50a28d00eb68ccd6617b4cc8950eb07","observation_id":"6e2b32c4-466f-4a0d-a84f-f9d077ad07d9","resolution":{"observed_at":"2026-08-07T15:39:37.161671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:36.821782Z","title":"Cumo: Scaling multimodal llm with co-upcycled mixture-of-experts,","venue":null,"work_id":"b342623f-b7a8-46d2-9a0e-4eb070742817","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:32.629949Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:a5c68d429cff959b83fe64809a1c90e1987106e806cf8276728644fc9198ac47","observation_id":"160b6d9e-fe8d-43f6-85ae-7b74809bd176","resolution":{"observed_at":"2026-08-07T15:39:36.979109Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:36.673395Z","title":"Chartmoe: Mixture of expert connector for advanced chart understanding,","venue":null,"work_id":"388bd134-7d33-41bd-aa0c-1f395460b59f","year":2025},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:32.719550Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:bba6b0614cccafc01fc5b28ff971531eb4a1252d9ca786b82a4feff7fc4fc117","observation_id":"e513e606-4937-4e6c-a932-ad186f76fced","resolution":{"observed_at":"2026-08-07T15:39:36.730999Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:36.447771Z","title":"Dense connector for mllms,","venue":null,"work_id":"5ea14266-209f-4c15-a5c6-49714ec4d3ae","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:32.804638Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:468303620a67300027e069707bb59451d0b7dcb4cb4e6de8549c11f14f94a347","observation_id":"34b72aa8-89c5-4db9-b9c7-da0fbd3ffdd5","resolution":{"observed_at":"2026-08-07T15:39:36.560704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:36.318265Z","title":"Visual instruction tuning,","venue":null,"work_id":"ccabc5a9-1f9c-4826-a281-4d8ef6da491c","year":2023},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:32.910843Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:e5a111c6df66c130d43f2df0dfbc508b10ee786d100338b007507fd850a1cd0c","observation_id":"1884dd08-84ba-4140-bcec-d54b4ef10f95","resolution":{"observed_at":"2026-08-07T15:39:36.381234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:36.133869Z","title":"Vila: On pre-training for visual language models,","venue":null,"work_id":"d759e695-6c55-48e5-aeff-c0f525b84ad8","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.015404Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:aff5ae649f7242e44a7162f88ffda1eda0a379f199559b9012310e1fce3ce059","observation_id":"5b7540b4-6bb1-47d6-903e-38d789347c5d","resolution":{"observed_at":"2026-08-07T15:39:36.213955Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14402","last_updated":"2024-11-21T18:31:25Z","snapshot_observed_at":"2026-08-12T15:11:04.635091Z","submitted_at":"2024-11-21T18:31:25Z","title":"Multimodal Autoregressive Pre-training of Large Vision Encoders","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14402","snapshot_observed_at":"2026-08-07T15:39:33.091881Z","title":"Multimodal autoregressive pre-training of large vi- sion encoders,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.091881Z"},"links":{"cited_paper":"/paper/2411.14402","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:45a33d2c10abe5024542387dd97733d042ddb289a24d67d597b5cc06a338d772","observation_id":"c8812366-15d3-45be-8a37-11d69eec5b6e","resolution":{"observed_at":"2026-08-07T15:39:33.091881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:35.847311Z","title":"Meteor: Mamba-based traversal of rationale for large language and vision models,","venue":null,"work_id":"71a74906-b4e2-4232-a2c7-4c2c45e5d1a9","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.156800Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:37022b2d56616a7369b113d3ff925240d6427a332fb43f00439db8eba00b8997","observation_id":"dbefa5de-ac70-46fe-8edc-5698f6e910c5","resolution":{"observed_at":"2026-08-07T15:39:35.958993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:35.645083Z","title":"Lora: Low-rank adaptation of large language mod- els,","venue":null,"work_id":"b626586e-1fb4-4a9d-b4b4-7bdccde6697a","year":2021},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.253776Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:1e8803aba451a1b33713ad42259e3c6bdcfe8e033b766420b7bbfab712498554","observation_id":"a04070d8-7a4a-48f0-aaea-7ca62d1dd6ca","resolution":{"observed_at":"2026-08-07T15:39:35.751900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.08906","last_updated":"2022-04-29T23:24:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-02-17T21:39:10Z","title":"ST-MoE: Designing Stable and Transferable Sparse Expert Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.08906","snapshot_observed_at":"2026-08-07T15:39:33.358907Z","title":"St-moe: Designing stable and transferable sparse expert models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.358907Z"},"links":{"cited_paper":"/paper/2202.08906","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:ad905aedf62891c9d08f7c98226d323f3089cfe6a51a83d5ec66f1d11d0e8550","observation_id":"c76be01e-d01f-4fa6-9cc7-8aa6b74ce86f","resolution":{"observed_at":"2026-08-07T15:39:33.358907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11829","last_updated":"2024-10-15T17:55:22Z","snapshot_observed_at":"2026-08-12T22:22:29.698348Z","submitted_at":"2024-10-15T17:55:22Z","title":"MMFuser: Multimodal Multi-Layer Feature Fuser for Fine-Grained Vision-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11829","snapshot_observed_at":"2026-08-07T15:39:33.418185Z","title":"Mmfuser: Multimodal multi-layer feature fuser for fine-grained vision-language understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.418185Z"},"links":{"cited_paper":"/paper/2410.11829","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:253e5f76b218580cecc81f9b78a856b069c323c8f246d58afe16498f41608d80","observation_id":"4a45b99a-6465-4660-92f8-72b2677fe431","resolution":{"observed_at":"2026-08-07T15:39:33.418185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.00496","last_updated":"2018-10-28T14:29:46Z","snapshot_observed_at":"2026-07-06T06:58:51.877372Z","submitted_at":"2018-09-03T08:38:34Z","title":"LRS3-TED: a large-scale dataset for visual speech recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.00496","snapshot_observed_at":"2026-08-07T15:39:33.522351Z","title":"Lrs3-ted: a large-scale dataset for visual speech recognition,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.522351Z"},"links":{"cited_paper":"/paper/1809.00496","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:65f80db615bc77630aee93d83126d741daaaebfd791d3c4192085da6dd74082e","observation_id":"aae50937-c189-4c57-afcb-9d96beb8aa9e","resolution":{"observed_at":"2026-08-07T15:39:33.522351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:35.465232Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":"cc668bf2-fac0-4aa7-89d2-30eaa33dae03","year":2023},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.634650Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:b1bb77056004d68a1c8ed232cedc894991338ff1a71c88605cc4b3e5ae6923b6","observation_id":"26a5e078-3f87-49dc-bd75-d48bb2d935d7","resolution":{"observed_at":"2026-08-07T15:39:35.548830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-10T16:40:37.411115Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T15:39:33.700092Z","title":"The llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.700092Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:06895f2cb9f944112dd02681c95cd4b93b305f04b1d400223865cb52d7aabbd5","observation_id":"cf986061-082d-400a-9f3a-ca53fa4dd693","resolution":{"observed_at":"2026-08-07T15:39:33.700092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:35.293774Z","title":"Towards a unified view of parameter-efficient transfer learning,","venue":null,"work_id":"94016646-eac6-4aa4-b777-dcfcfe9a716c","year":2022},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.787501Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:b1f81e95281a5d85a67e430ca0d6ae9f2f24e731c1c229399cd2b1cf4c75d3cb","observation_id":"6916c243-6717-40fb-86ef-3cb2febef3aa","resolution":{"observed_at":"2026-08-07T15:39:35.376169Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:35.087097Z","title":"Parameter-efficient transfer learning of au- dio spectrogram transformers,","venue":null,"work_id":"db96f0be-db3a-441d-9670-fce7cd52397e","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.881568Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:828e8b48317cd25fa6579dbabdd1d0132b886c2f8569fcad50f1c9de0c344130","observation_id":"938640d8-9e79-4b54-b176-dde2deed5f17","resolution":{"observed_at":"2026-08-07T15:39:35.200745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.06666","last_updated":"2025-03-01T12:59:49Z","snapshot_observed_at":"2026-08-12T22:47:34.452271Z","submitted_at":"2024-09-10T17:34:34Z","title":"LLaMA-Omni: Seamless Speech Interaction with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.06666","snapshot_observed_at":"2026-08-07T15:39:33.966621Z","title":"Llama-omni: Seamless speech interaction with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:33.966621Z"},"links":{"cited_paper":"/paper/2409.06666","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:8c90e83a413377c9621eabeae11171f7191f43eae3323e97ceb91f07a3c57fe7","observation_id":"c07ba69b-84a4-4031-a186-ef95e4cb43fe","resolution":{"observed_at":"2026-08-07T15:39:33.966621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:39:34.914229Z","title":"Mixtures of experts for audio-visual learning,","venue":null,"work_id":"dad7a06e-c85d-45a2-8a0c-07ef04821aa8","year":2024},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:34.034276Z"},"links":{"citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:ca388c2df631144c393596c276a659c6a176fd572eea3f2f6f3437fb4eb29839","observation_id":"088c4cf1-edc1-41ea-9160-20f5dfcdbfe9","resolution":{"observed_at":"2026-08-07T15:39:34.993972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10447","last_updated":"2025-05-21T04:07:33Z","snapshot_observed_at":"2026-08-09T14:12:28.614395Z","submitted_at":"2025-02-11T11:01:05Z","title":"MoHAVE: Mixture of Hierarchical Audio-Visual Experts for Robust Speech Recognition","version":2},"cited_work":{"arxiv_id":"2502.10447","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.10447","snapshot_observed_at":"2026-08-07T15:39:34.251185Z","title":"MoHAVE: Mixture of Hierarchical Audio-Visual Experts for Robust Speech Recognition","venue":"eess.AS","work_id":"ac49cdf8-dc2d-4103-aa07-c6e714ba7f87","year":2025},"citing_paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T15:39:34.111978Z"},"links":{"cited_paper":"/paper/2502.10447","citing_paper":"/paper/2505.14336"},"observation_digest":"sha256:61254361f5f90343ed1f0613c5df3a29ed1133f2745cc6697c7b5360fb703339","observation_id":"10189419-0fe1-4ad1-9f36-c209cea88af3","resolution":{"observed_at":"2026-08-07T15:39:34.345367Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.14336","last_updated":"2025-05-21T14:22:18Z","latest_version":2,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-09T14:12:04.052166Z","submitted_at":"2025-05-20T13:20:55Z","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach"},"reference_resolution":{"displayed":62,"state_counts":{"malformed_identifier":1,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":14,"verified_exact":2,"verified_fuzzy":44},"total_outbound_references":62},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 62 of 62 outbound references and 1 inbound Pith citation observation for arXiv:2505.14336."}