{"as_of":"2026-08-23T08:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8431d9a90acbf875ce5aea23dcba3e684c9d31fcaee5cfd9554ad82414007c6e","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:24:35.057623Z","state":"measured"},{"denominator":46,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":46,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:24:32.537583Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-07T14:53:55.827006Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19037","snapshot_observed_at":"2026-08-07T14:24:32.537583Z","title":"Speech-aware lan- guage models (SLMs) [9–18] extend this success by integrating speech perception into the broad textual knowledge of LLMs","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:32.537583Z"},"links":{"cited_paper":"/paper/2505.19037","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:d1b42f70a3429c8933db1f6284efbdf184bb863c65744f4fb9e17ef8f649a8b9","observation_id":"2a6b7b26-a6d3-4ab5-961c-069dd69f4cf5","resolution":{"observed_at":"2026-08-07T14:24:32.537583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19037","snapshot_observed_at":"2026-08-07T05:44:08.447266Z","title":"Speech-ifeval: Evaluating instruction-following and quantifying catastrophic forgetting in speech- aware language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.07233","last_updated":"2025-09-12T18:49:28Z","snapshot_observed_at":"2026-08-18T02:56:33.217331Z","submitted_at":"2025-06-08T17:36:50Z","title":"Reducing Object Hallucination in Large Audio-Language Models via Audio-Aware Decoding","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:08.447266Z"},"links":{"cited_paper":"/paper/2505.19037","citing_paper":"/paper/2506.07233"},"observation_digest":"sha256:ce65501247a03ff801738b710db35afb2cd791843b4db237e4a6129b8081876a","observation_id":"b1951685-0dca-4ca2-a233-c7ecd738c91f","resolution":{"observed_at":"2026-08-07T05:44:08.447266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19037","snapshot_observed_at":"2026-08-07T10:17:47.081584Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.10019","last_updated":"2025-06-06T11:09:46Z","snapshot_observed_at":"2026-08-17T19:45:22.860227Z","submitted_at":"2025-06-06T11:09:46Z","title":"A Survey of Automatic Evaluation Methods on Text, Visual and Speech Generations","version":1},"reference_index":234,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:47.081584Z"},"links":{"cited_paper":"/paper/2505.19037","citing_paper":"/paper/2506.10019"},"observation_digest":"sha256:9cc1d0d216bd9ebe47740cf8477c46dc36bdafdf4522c889a89188162dfa30df","observation_id":"73550c0b-7382-42e0-916b-836d52fb3f4c","resolution":{"observed_at":"2026-08-07T10:17:47.081584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"cited_work":{"arxiv_id":"2505.19037","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.19037","snapshot_observed_at":"2026-07-07T14:53:55.827006Z","title":"Speech-ifeval: Evaluating instruction-following and quantifying catastrophic forgetting in speech- aware language models","venue":"eess.AS","work_id":"486d8c9b-c1eb-4f21-8da6-20a0d8e2e34b","year":2025},"citing_paper":{"arxiv_id":"2605.21008","last_updated":"2026-05-20T10:44:56Z","snapshot_observed_at":"2026-08-15T09:30:05.500365Z","submitted_at":"2026-05-20T10:44:56Z","title":"A Survey of Audio Reasoning in Multimodal Foundation Models","version":1},"reference_index":117,"source":"pdf_text","source_observed_at":"2026-05-21T02:08:06.976461Z"},"links":{"cited_paper":"/paper/2505.19037","citing_paper":"/paper/2605.21008"},"observation_digest":"sha256:5980eda1c5b4ad9c81440c52ba02f01d1d6481a8f8efbfe7588b84b79960636e","observation_id":"02887523-90a9-4bb8-bad3-a48d717f4ffe","resolution":{"observed_at":"2026-05-21T02:09:24.086329Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"cited_work":{"arxiv_id":"2505.19037","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.19037","snapshot_observed_at":"2026-07-07T14:53:55.827006Z","title":"Speech-ifeval: Evaluating instruction-following and quantifying catastrophic forgetting in speech- aware language models","venue":"eess.AS","work_id":"486d8c9b-c1eb-4f21-8da6-20a0d8e2e34b","year":2025},"citing_paper":{"arxiv_id":"2606.29534","last_updated":"2026-06-28T17:57:41Z","snapshot_observed_at":"2026-08-12T01:50:59.406395Z","submitted_at":"2026-06-28T17:57:41Z","title":"Preference-ASR: A Preference-Aware Test Set for Benchmarking ASR in the Era of Speech LLMs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-30T07:26:38.380118Z"},"links":{"cited_paper":"/paper/2505.19037","citing_paper":"/paper/2606.29534"},"observation_digest":"sha256:c25b0667bee6042ed43834e401c776b60f8556d188977a8598c2c50598d45da3","observation_id":"4b5dee97-94de-4cce-8e2c-254715369ddf","resolution":{"observed_at":"2026-06-30T07:34:21.796509Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"cited_work":{"arxiv_id":"2505.19037","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.19037","snapshot_observed_at":"2026-07-07T14:53:55.827006Z","title":"Speech-ifeval: Evaluating instruction-following and quantifying catastrophic forgetting in speech- aware language models","venue":"eess.AS","work_id":"486d8c9b-c1eb-4f21-8da6-20a0d8e2e34b","year":2025},"citing_paper":{"arxiv_id":"2607.05364","last_updated":"2026-07-15T17:16:51Z","snapshot_observed_at":"2026-08-15T21:16:03.653789Z","submitted_at":"2026-07-06T17:40:54Z","title":"REDDIT: Correcting Model-Generated Timestamp Drift in ASR without Forgetting via Replay-Based Distribution Editing","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-07T14:52:35.127533Z"},"links":{"cited_paper":"/paper/2505.19037","citing_paper":"/paper/2607.05364"},"observation_digest":"sha256:bdd17188f927366d2ab29c5c5af6f68df8359f7a8a00108da04425173b58c344","observation_id":"963c8688-4e22-42df-96b9-f4e92dd0db1e","resolution":{"observed_at":"2026-07-07T14:53:55.828890Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19037","snapshot_observed_at":"2026-08-02T08:34:04.723035Z","title":"Speech-IFEval: Evaluating instruction-following and quantifying catastrophic forgetting in speech- aware language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.05364","last_updated":"2026-07-15T17:16:51Z","snapshot_observed_at":"2026-08-15T21:16:03.653789Z","submitted_at":"2026-07-06T17:40:54Z","title":"REDDIT: Correcting Model-Generated Timestamp Drift in ASR without Forgetting via Replay-Based Distribution Editing","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-02T08:34:04.723035Z"},"links":{"cited_paper":"/paper/2505.19037","citing_paper":"/paper/2607.05364"},"observation_digest":"sha256:dfb51d039e3d91b3f4158db048a8472084373c7d8618cc38aea05942af42a996","observation_id":"f2bd38d6-64e9-4c8f-9144-a718b81b1678","resolution":{"observed_at":"2026-08-02T08:34:04.723035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.19037/citation-record","integrity":"/paper/2505.19037/integrity","json":"/paper/2505.19037/citation-record.json","paper":"/paper/2505.19037"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19037","snapshot_observed_at":"2026-08-07T14:24:32.537583Z","title":"Speech-aware lan- guage models (SLMs) [9–18] extend this success by integrating speech perception into the broad textual knowledge of LLMs","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:32.537583Z"},"links":{"cited_paper":"/paper/2505.19037","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:d1b42f70a3429c8933db1f6284efbdf184bb863c65744f4fb9e17ef8f649a8b9","observation_id":"2a6b7b26-a6d3-4ab5-961c-069dd69f4cf5","resolution":{"observed_at":"2026-08-07T14:24:32.537583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:37.562897Z","title":"To assess their instruction-following capabilities, several bench- marks have been developed [27–29]","venue":null,"work_id":"b75cc3d7-8f7f-48f6-bb82-d0d3a40635b2","year":null},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:32.606279Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:27bd786cb798fb427aed9a9ac876c765c2b585dce8f62346a82353a9e29a9b8b","observation_id":"645f1457-e873-4d0c-abd9-065f813b9006","resolution":{"observed_at":"2026-08-07T14:24:37.568904Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:37.545249Z","title":null,"venue":null,"work_id":"69fb7946-3124-46a6-a4fc-eefc7b2a3255","year":null},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:32.692638Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:f49102df9b55feb2223b15198d4462d69f270099aae3dc6edea4313fa0471c67","observation_id":"b79fd799-4ded-4b67-9ec4-f50950a2b1f6","resolution":{"observed_at":"2026-08-07T14:24:37.551140Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:37.525905Z","title":"Model details As shown in Table 3, we evaluate publicly available SLMs [9–11, 13–16] alongside their corresponding LLM components [2, 6, 7], which serve as reference systems","venue":null,"work_id":"aee39ba0-1f07-48a6-bf60-76078d78edd2","year":null},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:32.722816Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:9e7f0af4eb384801bb4cc58cd24641196cb59b16ec5b953751f3585c6a78b4ea","observation_id":"b0caaaed-1d0c-46bf-be43-0a9af3964cb2","resolution":{"observed_at":"2026-08-07T14:24:37.531817Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:37.505144Z","title":"Results on Speech-IFEval Table 2 presents a comprehensive evaluation of Speech-IFEval, focusing exclusively on instruction-following ability while dis- regarding speech perception","venue":null,"work_id":"4a1ac19d-9bbb-40c0-93b9-77a39d986c36","year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:32.757494Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:389e8ac55b075363a0b62265e9e6d51b126f03a0e9fad40048402a7d10aa2c10","observation_id":"20672c36-746a-4d22-abc8-4c6153f0fb13","resolution":{"observed_at":"2026-08-07T14:24:37.511838Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:37.488195Z","title":"This separation enables a more precise evaluation of whether and how SLMs retain their textual competencies af- ter speech-text training","venue":null,"work_id":"5bc1a9ba-d1f2-477a-b8e1-1ffb0605ec58","year":null},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:32.841787Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:afdde5210048c3872a1df2e0c7cd7710e73b89e1f3ca007e81071f2f878f2902","observation_id":"886d49cc-d10f-45f7-aafa-9bf14b787058","resolution":{"observed_at":"2026-08-07T14:24:37.494175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T14:24:32.959694Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:32.959694Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:0393ce324600ceb991db90052ea76e184a58047a34c4d9a278ee90d9b3dc9395","observation_id":"ed31a8db-fcc6-4ab7-8375-94b7b68b3dfe","resolution":{"observed_at":"2026-08-07T14:24:32.959694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-08-20T15:31:01.041088Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-07T14:24:33.034169Z","title":"Qwen technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.034169Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:d2d421fef73724d0ca4c10cbb33cc5b5898f4a41b7f8a9cc98030a64ba3e9df2","observation_id":"b3177d3d-c58e-4962-8df1-29832954970a","resolution":{"observed_at":"2026-08-07T14:24:33.034169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T14:24:33.197355Z","title":"Qwen2 technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.197355Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:39214e60d7c1513cdab5005471dd47b0bc481b63a3575d56192859b47f9ca655","observation_id":"7f97b3a4-e44b-484d-90ae-53fa2119c4e0","resolution":{"observed_at":"2026-08-07T14:24:33.197355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T14:24:33.272243Z","title":"Llama 2: Open foundation and fine-tuned chat models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.272243Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:b1f9f22d7c34aa966e557feeb7f0fb4f4dab1dfa4e303283c6b2b248c98b6f34","observation_id":"e0960d5c-1757-4510-807f-811fa14e015a","resolution":{"observed_at":"2026-08-07T14:24:33.272243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T14:24:33.346212Z","title":"The llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.346212Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:0432d24026468fca1b60ddb8776ab10fae26909d7f1d617510a2700590a5c5b4","observation_id":"b1b10236-a63c-468b-a77f-539b5803095f","resolution":{"observed_at":"2026-08-07T14:24:33.346212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:33.413497Z","title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.413497Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:830b1be0807023295fb76680bacecb0296180e2e00799629f632a9b15cfc7be2","observation_id":"3b2ad39f-88a2-4679-a9fe-e9404ce3d280","resolution":{"observed_at":"2026-08-07T14:24:33.413497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-08-20T18:27:04.837880Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-07T14:24:33.470714Z","title":"Gemini: a family of highly capable multimodal models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.470714Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:979df736bc4eaddcc58b4a0e3048f45c8c8a28312f1668715781911825259e74","observation_id":"39f79dbb-a3d9-48b2-bac6-14fe91ccb93e","resolution":{"observed_at":"2026-08-07T14:24:33.470714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07919","last_updated":"2023-12-21T10:20:42Z","snapshot_observed_at":"2026-08-07T10:17:55.688598Z","submitted_at":"2023-11-14T05:34:50Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07919","snapshot_observed_at":"2026-08-07T14:24:33.559481Z","title":"Qwen-audio: Advancing universal audio understand- ing via unified large-scale audio-language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.559481Z"},"links":{"cited_paper":"/paper/2311.07919","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:ae8df781236617b6e42bf924d5d7334e01be46cd14c566b75757edf0a0ec0037","observation_id":"bf30c3c8-2ba5-426a-9aa2-36a01b5e3467","resolution":{"observed_at":"2026-08-07T14:24:33.559481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-08-14T01:27:16.843576Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-07T14:24:33.631428Z","title":"Qwen2-audio technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.631428Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:7d8309f33c5abb30bbe571744fa2143d455c48797331f656aa7acd64bfa2547c","observation_id":"2a20b2b8-01bb-47b4-a429-7979f2e0a8a3","resolution":{"observed_at":"2026-08-07T14:24:33.631428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.20007","last_updated":"2025-01-27T08:46:09Z","snapshot_observed_at":"2026-08-22T06:27:55.863481Z","submitted_at":"2024-09-30T07:01:21Z","title":"DeSTA2: Developing Instruction-Following Speech Language Model Without Speech Instruction-Tuning Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.20007","snapshot_observed_at":"2026-08-07T14:24:33.707065Z","title":"Developing instruction- following speech language model without speech instruction- tuning data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.707065Z"},"links":{"cited_paper":"/paper/2409.20007","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:66138487c54520393b5aca3af6f133182e1d4e04e58a0332d6ce5eb6e557f0b8","observation_id":"0593cf21-cea0-43b3-aead-54ce8edbed39","resolution":{"observed_at":"2026-08-07T14:24:33.707065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:37.460447Z","title":"Desta: Enhancing speech language models through descriptive speech-text alignment,","venue":null,"work_id":"f3ea64b7-b250-46ee-b982-aa1a7f68aa4c","year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.750426Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:8952ad8bd798dc81c323ac934752c3f6147d62bac808c84c859975a9620692da","observation_id":"fe847266-4cc8-46a1-b98d-39bbdb5c14e8","resolution":{"observed_at":"2026-08-07T14:24:37.465571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:33.808780Z","title":"SALMONN: Towards generic hearing abilities for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.808780Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:abe85776f4c2665ac9873e789b648b014a848bf42790ac232b57a2007d9fc74f","observation_id":"49741d8f-c750-49ea-94fd-1277e830530b","resolution":{"observed_at":"2026-08-07T14:24:33.808780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:37.432980Z","title":"Joint audio and speech understanding,","venue":null,"work_id":"197413b8-bb3e-4767-b885-7e52d52ef87e","year":2023},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.870828Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:6b920c6af6a46eebe40f753ab671bd59303d81b0dd18de5a0353f0723253eed7","observation_id":"d81abbce-7f91-47df-b675-06c6dfdce61f","resolution":{"observed_at":"2026-08-07T14:24:37.438341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02678","last_updated":"2024-10-03T17:04:48Z","snapshot_observed_at":"2026-08-21T16:42:22.094560Z","submitted_at":"2024-10-03T17:04:48Z","title":"Distilling an End-to-End Voice Assistant Without Instruction Training Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02678","snapshot_observed_at":"2026-08-07T14:24:33.963171Z","title":"Dis- tilling an end-to-end voice assistant without instruction training data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:33.963171Z"},"links":{"cited_paper":"/paper/2410.02678","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:62c412cabac04cb029834eb281c80bb86f5570814cfd6c92bfefa1f29ec05aa6","observation_id":"9b55b015-bacf-4433-979d-63512062722d","resolution":{"observed_at":"2026-08-07T14:24:33.963171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.03872","last_updated":"2024-06-06T09:02:31Z","snapshot_observed_at":"2026-08-17T21:15:44.958706Z","submitted_at":"2024-06-06T09:02:31Z","title":"BLSP-Emo: Towards Empathetic Large Speech-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.03872","snapshot_observed_at":"2026-08-07T14:24:34.022294Z","title":"Blsp- emo: Towards empathetic large speech-language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.022294Z"},"links":{"cited_paper":"/paper/2406.03872","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:1376082fc13715a47eb95947596c07283600f07b40ea1d387c370dfc858d2c8c","observation_id":"71b93392-4140-458c-b4ae-827014a5d997","resolution":{"observed_at":"2026-08-07T14:24:34.022294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.00656","last_updated":"2024-09-21T15:27:30Z","snapshot_observed_at":"2026-08-19T23:59:29.658982Z","submitted_at":"2024-03-31T12:01:32Z","title":"WavLLM: Towards Robust and Adaptive Speech Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.00656","snapshot_observed_at":"2026-08-07T14:24:34.067293Z","title":"Wavllm: Towards robust and adaptive speech large language model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.067293Z"},"links":{"cited_paper":"/paper/2404.00656","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:8e14b08123636b247269870ef688d61ae8a3e864176b169606f7acdf0280ffb0","observation_id":"39dc1cdd-a81c-44c2-89b0-8999099d6d37","resolution":{"observed_at":"2026-08-07T14:24:34.067293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:34.101553Z","title":"Speech-copilot: Leveraging large language models for speech processing via task decomposition, modularization, and program generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.101553Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:152698e87352aedbfc8f6f38ff465ccc7c27ac3fff0605ffedc0565734ef15a9","observation_id":"388bbf6e-702d-4033-a8b4-9560f4d0f11b","resolution":{"observed_at":"2026-08-07T14:24:34.101553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:37.407464Z","title":"Dynamic-superb: Towards a dynamic, col- laborative, and comprehensive instruction-tuning benchmark for speech,","venue":null,"work_id":"7bd93c56-a92a-4ce3-bc3b-09bff5768740","year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.176459Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:e811db4570267225b6697f6df5211440ab9f388818217e79940fb32d590d4b45","observation_id":"b2e998b2-2c2f-400a-9e0b-91183918f54f","resolution":{"observed_at":"2026-08-07T14:24:37.412066Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:37.391696Z","title":"AIR-bench: Benchmarking large audio-language models via generative comprehension,","venue":null,"work_id":"65c986f2-8dc8-42c9-be1c-34187672e720","year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.249708Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:a0a56f43583f6498ef7e2f44a7f8369e8dcc8c17944894ed7136e5583962e0a2","observation_id":"1b177bc9-34fe-4da1-b978-29237c697713","resolution":{"observed_at":"2026-08-07T14:24:37.396637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:37.375040Z","title":"Dynamic-SUPERB phase-2: A collabora- tively expanding benchmark for measuring the capabilities of spo- ken language models with 180 tasks,","venue":null,"work_id":"d414f0f5-1772-4ced-b8e0-fe82ff1383ac","year":2025},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.318244Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:fb11150149eb0020c1d3fd0b828df9682b144f9c88c2a51f1b35b99f253a8761","observation_id":"731379bd-e6a0-4014-bfab-9508deb50827","resolution":{"observed_at":"2026-08-07T14:24:37.380361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:37.207418Z","title":"Beyond single-audio: Advancing multi-audio processing in audio large language models,","venue":null,"work_id":"1d3d87f2-5322-4222-849f-a075240f9934","year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.366953Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:c99683d4d7f1e213a995e6d884f8cf3981846be4b5bc6d87cc44f5dd25a241e5","observation_id":"09c4d480-1e0c-4b01-b5e6-24f7f22a6f04","resolution":{"observed_at":"2026-08-07T14:24:37.297798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05167","last_updated":"2025-07-28T15:07:08Z","snapshot_observed_at":"2026-08-12T11:06:15.857670Z","submitted_at":"2024-12-06T16:34:15Z","title":"Benchmarking Open-ended Audio Dialogue Understanding for Large Audio-Language Models","version":2},"cited_work":{"arxiv_id":"2412.05167","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.05167","snapshot_observed_at":"2026-08-07T14:24:35.251762Z","title":"Benchmarking Open-ended Audio Dialogue Understanding for Large Audio-Language Models","venue":"cs.AI","work_id":"a3d3fde6-d1d6-49c5-846b-3f8d64534a6b","year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.431110Z"},"links":{"cited_paper":"/paper/2412.05167","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:35bb10bb0e1b003084ae392c904209154066baa582b82ad86130b23e02abacd7","observation_id":"de478c3f-6be0-4767-bfbd-9d0a3fde9c88","resolution":{"observed_at":"2026-08-07T14:24:35.328214Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:36.902812Z","title":"MMAU: A mas- sive multi-task audio understanding and reasoning benchmark,","venue":null,"work_id":"96ca096f-c782-44e8-985d-64b9354cd247","year":2025},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.486517Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:9970c20304d67e767aa24deb31e21c6ca158990f8d7505600519efcabe2eae41","observation_id":"3cce925a-6056-447c-98bd-937c24735c65","resolution":{"observed_at":"2026-08-07T14:24:37.106691Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1312.6211","last_updated":"2015-03-04T01:43:31Z","snapshot_observed_at":"2026-08-14T23:50:42.566579Z","submitted_at":"2013-12-21T06:31:41Z","title":"An Empirical Investigation of Catastrophic Forgetting in Gradient-Based Neural Networks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1312.6211","snapshot_observed_at":"2026-08-07T14:24:34.571005Z","title":"An empirical investigation of catastrophic forgetting in gradient- based neural networks,","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.571005Z"},"links":{"cited_paper":"/paper/1312.6211","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:3a5fcd999fa244d947a7f7433c17f96aaf67b655d760a84b3c1374835ba30afe","observation_id":"91d20702-f918-456a-839c-3d382f463706","resolution":{"observed_at":"2026-08-07T14:24:34.571005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:36.790559Z","title":"Self- powered LLM modality expansion for large speech-text models,","venue":null,"work_id":"f13a63e3-ce22-45c5-af30-402ec6fcf29e","year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.631664Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:14dbbcc12a6522915ce2b2a6c5e10e8f6534241a38c51c708a75f3dc1a1516d2","observation_id":"8c2aa1b4-b780-47e3-a703-5d15781dc6ba","resolution":{"observed_at":"2026-08-07T14:24:36.840917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-07T14:24:34.667540Z","title":"Instruction-following evaluation for large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.667540Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:517d3e03e865dc029c2c146569bc4aa38cbbb13ead408fb9650a00b9d8daa1d0","observation_id":"4658673b-028a-4740-b015-d0fcb7c79845","resolution":{"observed_at":"2026-08-07T14:24:34.667540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:36.611005Z","title":"InFoBench: Evaluating instruction following ability in large language models,","venue":null,"work_id":"5c26d3ba-d762-46e6-bfa5-b3bee51f2d8a","year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.709219Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:41590de6586cb06fe3d1724ce4b7141bb4bcbc1a00925a43ea7de08f83452b74","observation_id":"500d6a1e-db70-4441-bb02-518cb789caa2","resolution":{"observed_at":"2026-08-07T14:24:36.691726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:36.485820Z","title":"FOFO: A benchmark to evaluate LLMs’ format- following capability,","venue":null,"work_id":"59bb2a69-c256-454a-a3ba-0b4666a4bb05","year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.769578Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:aa2474856a35f0a3475fdc634a6d48b847d3a84fe3f54dd6ae1bf3dc588a1f53","observation_id":"4c43b894-71b9-40e9-8346-da0bbdde1cff","resolution":{"observed_at":"2026-08-07T14:24:36.515256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:34.865350Z","title":"Audiobench: A universal benchmark for audio large language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.865350Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:2f716b0d5ef3a62e1587aecf8286c6b20859ca387abc86bf86d51988f7daee06","observation_id":"4978150e-c95e-4ebd-a420-3180ce5c811c","resolution":{"observed_at":"2026-08-07T14:24:34.865350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:36.260256Z","title":"Can large language models be an alternative to human evaluations?","venue":null,"work_id":"912852f5-ea76-41ab-91d8-c8d5fc4a7473","year":2023},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.930392Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:89ee1d9498d87249c5366f4e7d76f618665c9d7fcbdede1551014bb389aed60e","observation_id":"d78f16a3-78e7-42bd-a8b8-854def2df714","resolution":{"observed_at":"2026-08-07T14:24:36.357419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:36.156168Z","title":"Chain of thought prompting elicits reasoning in large language models,","venue":null,"work_id":"9dc7689f-0dfc-497c-a1e9-d43628ae82ac","year":2022},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:34.967697Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:132832902986945239ba516c87ff6eecb405ef8f910e281c0d14412b9ac65cd0","observation_id":"f273a5f7-3a38-41ae-87b4-05f016f38ec7","resolution":{"observed_at":"2026-08-07T14:24:36.205126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:35.901714Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":"9fa35aec-e794-4f50-a360-4e5bea0412c4","year":2023},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:35.003064Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:fc1e5b4afc0442ebedf7b56d1fd46fe711f5871d913c06e89fb62ef4bee6ad21","observation_id":"1be62cdc-ea16-423e-997e-5f9a5c1f4106","resolution":{"observed_at":"2026-08-07T14:24:36.045834Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:35.712957Z","title":"emotion2vec: Self-supervised pre-training for speech emotion representation,","venue":null,"work_id":"65b4322d-a851-4aab-a213-fce6506a47e6","year":2024},"citing_paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:35.057623Z"},"links":{"citing_paper":"/paper/2505.19037"},"observation_digest":"sha256:96653060b09fb2ae75ad3273c81c438ec6009c53ae80a45dcb62685485401e55","observation_id":"6576f5f4-121c-4680-8a4c-92503297dc82","resolution":{"observed_at":"2026-08-07T14:24:35.834786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.19037","last_updated":"2025-05-25T08:37:55Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-17T12:03:52.615757Z","submitted_at":"2025-05-25T08:37:55Z","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":19,"verified_exact":1,"verified_fuzzy":17},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 7 inbound Pith citation observations for arXiv:2505.19037."}