{"id":"CVE-2026-34760","aliases":["GHSA-6c4r-fmh3-7rh8","PYSEC-2026-2299"],"url":"https://o3.security/vulnerability/CVE-2026-34760","summary":"vLLM: Downmix Implementation Differences as Attack Vectors Against Audio AI Models","details":"vLLM is an inference and serving engine for large language models (LLMs). From version 0.5.5 to before version 0.18.0, Librosa defaults to using numpy.mean for mono downmixing (to_mono), while the international standard ITU-R BS.775-4 specifies a weighted downmixing algorithm. This discrepancy results in inconsistency between audio heard by humans (e.g., through headphones/regular speakers) and audio processed by AI models (Which infra via Librosa, such as vllm, transformer). This issue has been patched in version 0.18.0.","published":"2026-04-02T18:59:49.638Z","modified":"2026-08-12T03:51:42.960050569Z","cvss":{"score":5.9,"severity":"MEDIUM","vector":"CVSS:3.1/AV:N/AC:H/PR:L/UI:N/S:U/C:N/I:H/A:L"},"epss":{"score":0.00267,"percentile":0.18594,"asOf":"2026-09-06"},"cisaKev":null,"exploitsKnown":0,"affectedPackages":[{"ecosystem":"PyPI","name":"vllm","fixedVersion":"0.18.0"}],"fix":{"url":"https://github.com/vllm-project/vllm/commit/c7f98b4d0a63b32ed939e2b6dfaa8a626e9b46c4","label":"vllm-project/vllm@c7f98b4"},"references":[{"type":"WEB","url":"https://github.com/vllm-project/vllm/releases/tag/v0.18.0"},{"type":"ADVISORY","url":"https://github.com/CVEProject/cvelistV5/tree/main/cves/2026/34xxx/CVE-2026-34760.json"},{"type":"ADVISORY","url":"https://github.com/vllm-project/vllm/security/advisories/GHSA-6c4r-fmh3-7rh8"},{"type":"ADVISORY","url":"https://nvd.nist.gov/vuln/detail/CVE-2026-34760"},{"type":"FIX","url":"https://github.com/vllm-project/vllm/commit/c7f98b4d0a63b32ed939e2b6dfaa8a626e9b46c4"},{"type":"FIX","url":"https://github.com/vllm-project/vllm/pull/37058"}],"provenance":{"sources":["OSV.dev","FIRST.org (EPSS)"],"lastVerified":"2026-08-12T03:51:42.960050569Z"}}