Script 'mail_helper' called by obssrc Hello community, here is the log from the commit of package python-vllm for openSUSE:Factory checked in at 2026-09-23 14:34:53 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ Comparing /work/SRC/openSUSE:Factory/python-vllm (Old) and /work/SRC/openSUSE:Factory/.python-vllm.new.383539 (New) ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Package is "python-vllm" Wed Sep 23 14:34:53 2026 rev:10 rq:1379803 version:0.30.0 Changes: -------- --- /work/SRC/openSUSE:Factory/python-vllm/python-vllm.changes 2026-09-10 11:52:17.852871066 +0200 +++ /work/SRC/openSUSE:Factory/.python-vllm.new.383539/python-vllm.changes 2026-09-23 14:36:35.030398193 +0200 @@ -1,0 +2,62 @@ +Tue Sep 22 21:42:21 UTC 2026 - Martin Pluskal <[email protected]> + +- Update to version 0.30.0: + * New DeepSeek-V4 CPU backend kernels and AMX-FP8 attention + kernels for Diamond Rapids CPUs + * Validation error responses are truncated, closing a large + response amplification from a small request body + * prompt_embeds and the image/video/audio _embeds inputs are + rejected by declared dense size before densification, + closing a memory amplification DoS (the limit is + VLLM_MAX_EMBED_DECODE_BYTES) + * Per-request video sampling is capped for the GLMGA and + Qwen2/3-VL backends (640/768 frames, 30 fps) + * Scale-out endpoints are no longer registered by a plain + "vllm serve" (pass --enable-scale-out); their multimodal + features are now bounds-checked against the prompt before + the engine sees them + * Late-interaction (ColBERT-style) scoring keys its query cache + by a server-generated id instead of the client's X-Request-Id, + so a client can no longer overwrite another client's cached + query embeddings + * routed_experts_prompt_start is checked against the prompt + length at the frontend; with --enable-return-routed-experts + an out-of-range value from an API client tripped an engine + core assert and took the engine down + * Integers in vllm_xargs outside the 64-bit range are rejected + with a validation error; they used to fail only after the + request was registered, leaking its state in the API server + on every such request + * "vllm bench serve" redacts --header values (e.g. auth tokens) + when printing its arguments and saving results + * Dropped: VLLM_PREFIX_CACHE_RETENTION_INTERVAL, + VLLM_MM_HASHER_ALGORITHM, use_fp4_indexer_cache, and + support for GPTQ checkpoints with desc_act=True + * YaRN is aligned with Transformers: rope_type yarn, + deepseek_yarn and deepseek_llama_scaling no longer re-scale + max_model_len, so context length drops for some models + * Many more changes; see upstream's release notes +- CVE-2026-92365: rescanning the whole output for the reasoning + markers on every decode step makes the thinking_token_budget + state machine quadratic in the generated length, letting an + unauthenticated client degrade latency for co-tenant requests + sharing the sampler step (boo#1280982) + * vllm-thinking-budget-scan-cursor.patch +- Raise the python-huggingface-hub floor to 1.31.0 and the + python-openai floor to 2.25.0 to match upstream +- Keep python-mcp unversioned although upstream now asks for + mcp >= 2.0.0: Factory carries 1.28.1, and only the optional + "--tool-server" integration reads the SDK 2.x tool schema + * vllm-relax-cpu-requirements.patch: relax the mcp pin in the + wheel metadata too, so the installed dist-info does not + assert a requirement the RPM cannot satisfy +- Is not affected by CVE-2026-90713 (boo#1280450): the Rust + frontend holding the tiktoken vocabulary loader is not built +- Is not affected by CVE-2026-93841 (boo#1281905): the Triton + bincount kernel belongs to Model Runner V2, which requires + Triton and falls back to V1 when it is absent +- Is not affected by CVE-2026-93989 (boo#1281906): the + unbounded bad-words write is in that same Triton V2 sampler; + the shipped V1 path assigns a bounded per-row slice + +------------------------------------------------------------------- Old: ---- vllm-0.29.0.tar.gz New: ---- vllm-0.30.0.tar.gz vllm-thinking-budget-scan-cursor.patch ----------(New B)---------- New: sharing the sampler step (boo#1280982) * vllm-thinking-budget-scan-cursor.patch - Raise the python-huggingface-hub floor to 1.31.0 and the ----------(New E)---------- ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ Other differences: ------------------ ++++++ python-vllm.spec ++++++ --- /var/tmp/diff_new_pack.rCsIXs/_old 2026-09-23 14:36:37.658506861 +0200 +++ /var/tmp/diff_new_pack.rCsIXs/_new 2026-09-23 14:36:37.660506944 +0200 @@ -48,7 +48,7 @@ %define acl_version 52.6.0 %bcond_without libalternatives Name: python-vllm%{psuffix} -Version: 0.29.0 +Version: 0.30.0 Release: 0 Summary: A high-throughput and memory-efficient inference and serving engine for LLMs License: Apache-2.0 @@ -68,6 +68,8 @@ Patch0: vllm-relax-cpu-requirements.patch # PATCH-FIX-OPENSUSE vllm-cpu-disable-rust-frontend.patch -- do not build the ~575-crate Rust frontend (unvendorable offline; runtime-optional) Patch1: vllm-cpu-disable-rust-frontend.patch +# PATCH-FIX-UPSTREAM vllm-thinking-budget-scan-cursor.patch boo#1280982 [email protected] -- CVE-2026-92365: resume the thinking-budget marker search from a cursor so a generation is not quadratic in its own length +Patch2: vllm-thinking-budget-scan-cursor.patch BuildRequires: %{python_module Jinja2} # Runtime modules imported eagerly by "import vllm" (needed for the %%check smoke test). BuildRequires: %{python_module cachetools} @@ -119,7 +121,7 @@ Requires: python-einops Requires: python-fastapi >= 0.133.0 Requires: python-filelock >= 3.16.1 -Requires: python-huggingface-hub >= 1.28.0 +Requires: python-huggingface-hub >= 1.31.0 Requires: python-ijson Requires: python-jsonschema >= 4.23.0 Requires: python-lark >= 1.2.2 @@ -127,13 +129,18 @@ # One of the optional structured-output backends, LazyLoader-imported only when # a request selects it. Requires: python-lm-format-enforcer >= 0.11.3 +# Upstream asks for mcp >= 2.0.0 because 0.30.0 reads the SDK 2.x snake_case +# tool schema (Tool.input_schema, InitializeResult.server_info). Factory only +# carries 1.28.1, and an unsatisfiable Requires would leave this package +# uninstallable, so keep it unversioned: mcp is imported lazily and solely by +# the optional "--tool-server" integration, the one feature that needs 2.x. Requires: python-mcp Requires: python-mistral-common >= 1.11.6 Requires: python-model-hosting-container-standards >= 0.1.14 Requires: python-msgspec Requires: python-numba >= 0.65.0 Requires: python-numpy -Requires: python-openai >= 2.0.0 +Requires: python-openai >= 2.25.0 Requires: python-openai-harmony >= 0.0.3 Requires: python-opencv >= 4.13.0 Requires: python-opentelemetry-api >= 1.27.0 @@ -288,7 +295,7 @@ rm -f $sd/vllm/distributed/kv_transfer/disagg_prefill_workflow.jpg rm -f $sd/vllm/vllm_flash_attn/.gitkeep # These modules carry a #!/usr/bin/env python shebang but are imported, not run. -sed -i '1{/^#!/d}' $sd/vllm/entrypoints/grpc_server.py +sed -i '1{/^#!/d}' $sd/vllm/entrypoints/launchers/grpc_server.py sed -i '1{/^#!/d}' $sd/vllm/entrypoints/launchers/dp_supervisor.py %fdupes $sd } ++++++ vllm-0.29.0.tar.gz -> vllm-0.30.0.tar.gz ++++++ /work/SRC/openSUSE:Factory/python-vllm/vllm-0.29.0.tar.gz /work/SRC/openSUSE:Factory/.python-vllm.new.383539/vllm-0.30.0.tar.gz differ: char 5, line 1 ++++++ vllm-relax-cpu-requirements.patch ++++++ --- /var/tmp/diff_new_pack.rCsIXs/_old 2026-09-23 14:36:37.737510128 +0200 +++ /var/tmp/diff_new_pack.rCsIXs/_new 2026-09-23 14:36:37.740510252 +0200 @@ -57,4 +57,13 @@ cloudpickle # allows pickling lambda functions in model_executor/models/registry.py watchfiles # required for http server to monitor the updates of TLS files python-json-logger # Used by logging as per examples/features/logging_configuration.md +@@ -51,7 +51,7 @@ + openai-harmony >= 0.0.3 # Required for gpt-oss + anthropic >= 0.71.0 + model-hosting-container-standards >= 0.1.14, < 1.0.0 +-mcp >= 2.0.0, < 3.0.0 ++mcp + opentelemetry-sdk >= 1.27.0 + opentelemetry-api >= 1.27.0 + opentelemetry-exporter-otlp >= 1.27.0 ++++++ vllm-thinking-budget-scan-cursor.patch ++++++ From: Martin Pluskal <[email protected]> Subject: [PATCH] Bound the thinking-budget marker search with a resume cursor References: CVE-2026-92365, boo#1280982 The thinking_token_budget state machine looks for the reasoning start/end markers with _find_last_sequence_index(), a backwards linear scan. While a marker has not been seen yet the cached position stays -1, so _update_think_state re-runs that scan over the *entire* accumulated output on every decode step. scan_offset only moves when a thinking section completes, so it does not bound the common case: a request whose output simply never contains the markers, or one still inside a long thinking section, rescans everything each step. That makes a single generation O(L^2) in its own length, and the cost is paid inside the sampler step, which is shared and serialised for the whole batch -- so an unauthenticated client can set thinking_token_budget with a large max_tokens and degrade latency for every co-tenant request. Measured on the sampler path alone, 4000 decode steps inspect 16,004,000 candidate positions and spend 6.1 s. A miss proves no occurrence begins at or before the last position tested, so the next step only has to look at the tokens appended since (plus the len(marker)-1 tail a marker can straddle). Record that position per marker and resume from it: the search becomes O(len(marker)) per step, the same 4000 steps inspect 8000 positions in 47 ms, and the resulting state is unchanged -- the scan still returns the last occurrence, because there is provably none earlier. The cursor is clamped to scan_offset, so completing a thinking section still invalidates it, and a shorter output (rejected speculative tokens) falls back to a full scan of the current section. --- a/vllm/v1/sample/thinking_budget_state.py +++ b/vllm/v1/sample/thinking_budget_state.py @@ -168,6 +168,37 @@ return i return -1 + def _scan_for_marker( + self, state: dict[str, Any], cursor_key: str, token_ids: list[int] + ) -> int: + """Last index of ``token_ids`` in ``output_tok_ids``, resuming a cursor. + + Only reached while the cached position is still -1, and a miss proves no + occurrence begins at or before the last position already tested. Storing + that position lets the next step look only at the tokens appended since, + so the marker search costs O(len(token_ids)) per decode step instead of + rescanning the whole output -- which is what made a single long + generation quadratic in its own length and let one client stall every + co-tenant request sharing the sampler step. + """ + if not token_ids: + return -1 + output = state.get("output_tok_ids", []) + scan_offset = state.get("scan_offset", 0) + # Highest index a match could start at, plus one. + resume_limit = len(output) - len(token_ids) + 1 + start = max(state.get(cursor_key, 0), scan_offset) + if start > resume_limit: + # Output is shorter than when the cursor was stored (rejected + # speculative tokens); the cursor is stale, rescan the section. + start = scan_offset + if start < resume_limit: + found = self._find_last_sequence_index(output[start:], token_ids) + if found >= 0: + return found + start + state[cursor_key] = max(start, resume_limit) + return -1 + def _init_state_entry( self, prompt_tok_ids: list[int] | None, thinking_token_budget: int ) -> dict[str, Any]: @@ -225,6 +256,8 @@ "bonus_token_forced": False, "continue_thinking": continue_thinking, "scan_offset": 0, + "start_scan_pos": 0, + "end_scan_pos": 0, } def _update_think_state(self, state: dict[str, Any]) -> None: @@ -237,23 +270,13 @@ return if state["start_thinking"] == -1: - scan_offset = state.get("scan_offset", 0) - output_slice = state.get("output_tok_ids", [])[scan_offset:] - start_thinking = self._find_last_sequence_index( - output_slice, self.think_start_token_ids + state["start_thinking"] = self._scan_for_marker( + state, "start_scan_pos", self.think_start_token_ids ) - if start_thinking >= 0: - start_thinking += scan_offset - state["start_thinking"] = start_thinking if state["end_thinking"] == -1: - scan_offset = state.get("scan_offset", 0) - output_slice = state.get("output_tok_ids", [])[scan_offset:] - end_thinking = self._find_last_sequence_index( - output_slice, self.think_end_token_ids + state["end_thinking"] = self._scan_for_marker( + state, "end_scan_pos", self.think_end_token_ids ) - if end_thinking >= 0: - end_thinking += scan_offset - state["end_thinking"] = end_thinking if ( not state.get("in_end", False)
