From 624e79aaa6eef906fe680ea42a7a3dbb6751f1e6 Mon Sep 17 00:00:00 2001 From: BadStorm Date: Sun, 2 Aug 2026 18:53:14 +0200 Subject: [PATCH] Standardize llama.cpp env vars and binary handling, add vLLM env passthrough --- .../llamacpp/llamacpp-adreno.Containerfile | 6 +- .../llamacpp/llamacpp-embedding.container | 6 +- .../llamacpp/llamacpp-opencl.Containerfile | 6 +- .../llamacpp/llamacpp-rocm.Containerfile | 34 +- containers/llamacpp/llamacpp-rpc.container | 5 +- .../llamacpp/llamacpp-vulkan.Containerfile | 12 +- containers/llamacpp/llamacpp.container | 10 +- containers/vllm/check-model.py | 353 ++++++++++++++++++ containers/vllm/entrypoint.sh | 14 + containers/vllm/vllm.container | 9 +- 10 files changed, 423 insertions(+), 32 deletions(-) create mode 100755 containers/vllm/check-model.py diff --git a/containers/llamacpp/llamacpp-adreno.Containerfile b/containers/llamacpp/llamacpp-adreno.Containerfile index be68d15..8c290b7 100644 --- a/containers/llamacpp/llamacpp-adreno.Containerfile +++ b/containers/llamacpp/llamacpp-adreno.Containerfile @@ -45,13 +45,13 @@ RUN VERSION=$(curl -s https://api.github.com/repos/ggml-org/llama.cpp/releases/l && tar -xzf llama.tar.gz -C . --strip-components=1 \ && rm llama.tar.gz -RUN chmod +x /app/llama-server +RUN find /app -maxdepth 1 -type f -exec chmod +x {} \; WORKDIR /app ENV PATH=/app:$PATH ENV LD_LIBRARY_PATH=/app:/usr/lib/aarch64-linux-gnu:/usr/lib/arm-linux-gnueabihf:$LD_LIBRARY_PATH -ENV HF_HUB_ENABLE_HF_TRANSFER=1 +ENV HF_XET_HIGH_PERFORMANCE=1 # Adreno-specific environment variables ENV VK_ICD_FILENAMES=/usr/share/vulkan/icd.d/freedreno_icd.json ENV MESA_GL_VERSION_OVERRIDE=4.6 @@ -62,7 +62,7 @@ ENV LLAMA_ARG_HOST=0.0.0.0 ENV LLAMA_ARG_PORT=8090 ENV LLAMA_ARG_HF_REPO=unsloth/gemma-4-26B-A4B-it-GGUF:IQ2_M #ENV LLAMA_ARG_MMPROJ_URL= -ENV LLAMA_ARG_NO_MMAP=true +ENV LLAMA_ARG_LOAD_MODE=none ENV LLAMA_ARG_CTX_SIZE=120000 #ENV LLAMA_API_KEY="" diff --git a/containers/llamacpp/llamacpp-embedding.container b/containers/llamacpp/llamacpp-embedding.container index a23e6ac..fa62894 100644 --- a/containers/llamacpp/llamacpp-embedding.container +++ b/containers/llamacpp/llamacpp-embedding.container @@ -1,6 +1,6 @@ [Container] ContainerName=llamacpp-embedding -Image=localhost/llamacpp:vulkan-amd64 +Image=localhost/llamacpp:vulkan Network=internal.network PublishPort=8091:8091 @@ -9,7 +9,7 @@ Volume=/srv/containers/aitools/models/hf:/root/.cache/huggingface/hub # ROCm / Vulkan — stessa GPU del container chat AddDevice=/dev/dri/renderD128 -PodmanArgs=--group-add=keep-groups --ipc=host +PodmanArgs=--group-add=keep-groups --ipc=host --pids-limit=-1 SecurityLabelType=container_runtime_t # Porta dedicata all'embedding @@ -23,7 +23,7 @@ Environment=LLAMA_ARG_HF_REPO=nomic-ai/nomic-embed-text-v1.5-GGUF:Q6_K # Flag fondamentale: avvia llama-server in modalità embedding-only Environment=LLAMA_ARG_EMBEDDING=true -Environment=LLAMA_ARG_NO_MMAP=true +Environment=LLAMA_ARG_LOAD_MODE=none # Contesto ridotto: gli embedding non hanno bisogno di 128k token Environment=LLAMA_ARG_CTX_SIZE=8192 diff --git a/containers/llamacpp/llamacpp-opencl.Containerfile b/containers/llamacpp/llamacpp-opencl.Containerfile index ef77c4f..49e6bfd 100644 --- a/containers/llamacpp/llamacpp-opencl.Containerfile +++ b/containers/llamacpp/llamacpp-opencl.Containerfile @@ -86,17 +86,17 @@ RUN cd llama.cpp \ RUN mkdir -p /app \ && find /build/llama.cpp/build/bin -maxdepth 1 -type f -exec cp {} /app/ \; \ && find /build/llama.cpp/build -maxdepth 3 -iname "*.so*" -exec cp -P {} /app/ \; \ - && chmod +x /app/llama-server \ + && find /app -maxdepth 1 -type f -exec chmod +x {} \; \ && rm -rf /build WORKDIR /app ENV PATH=/app:$PATH ENV LD_LIBRARY_PATH=/app:$LD_LIBRARY_PATH -ENV HF_HUB_ENABLE_HF_TRANSFER=1 +ENV HF_XET_HIGH_PERFORMANCE=1 ENV LLAMA_ARG_HOST=0.0.0.0 ENV LLAMA_ARG_PORT=8090 ENV LLAMA_ARG_HF_REPO=unsloth/Qwen3.5-35B-A3B-GGUF:Q2_K_XL -ENV LLAMA_ARG_NO_MMAP=true +ENV LLAMA_ARG_LOAD_MODE=none ENV LLAMA_ARG_CTX_SIZE=128000 # Adreno-specific: make sure Rusticl/OpenCL device is picked correctly ENV GGML_OPENCL_PLATFORM=0 diff --git a/containers/llamacpp/llamacpp-rocm.Containerfile b/containers/llamacpp/llamacpp-rocm.Containerfile index d5180f3..2801854 100644 --- a/containers/llamacpp/llamacpp-rocm.Containerfile +++ b/containers/llamacpp/llamacpp-rocm.Containerfile @@ -3,8 +3,8 @@ ### ### BUILD (gfx1151, default): ### podman build -t llamacpp:rocm -f llamacpp-rocm.Containerfile . -### BUILD (altro target): -### podman build --build-arg GPU_TARGET=gfx110X -t llamacpp:rocm -f llamacpp-rocm.Containerfile . +### BUILD (gfx1150): +### podman build --build-arg GPU_TARGET=gfx1150 -t llamacpp:rocm -f llamacpp-rocm.Containerfile . ### Export: podman save -o /home/badstorm/llamacpp-rocm.tar localhost/llamacpp:rocm FROM ubuntu:26.04 @@ -46,22 +46,40 @@ RUN curl -s https://api.github.com/repos/lemonade-sdk/llamacpp-rocm/releases/lat && unzip -q llama.zip -d /app \ && rm llama.zip /tmp/latest.json -RUN chmod +x /app/llama-server +# Lemonade e' spesso indietro di settimane: sovrascriviamo i binari/librerie llama.cpp con l'ultima release ufficiale, tenendo le librerie ROCm bundlate da lemonade (nessuna collisione di nomi). +RUN curl -s https://api.github.com/repos/ggml-org/llama.cpp/releases/latest -o /tmp/upstream.json \ + && UP_TAG=$(jq -r '.tag_name' /tmp/upstream.json) \ + && if [ -z "$UP_TAG" ] || [ "$UP_TAG" = "null" ]; then \ + echo "ERRORE: impossibile recuperare il tag latest di ggml-org/llama.cpp"; exit 1; \ + fi \ + && echo "Ultima versione ufficiale llama.cpp: $UP_TAG" \ + && UP_URL=$(jq -r '.assets[] | select(.name | test("^llama-.*-bin-ubuntu-rocm-[0-9.]+-x64\\.tar\\.gz$")) | .browser_download_url' /tmp/upstream.json | head -1) \ + && if [ -z "$UP_URL" ]; then \ + echo "ERRORE: nessun asset ubuntu-rocm nella release ${UP_TAG}"; \ + echo "Asset disponibili:"; jq -r '.assets[].name' /tmp/upstream.json; \ + exit 1; \ + fi \ + && echo "Scarico: $UP_URL" \ + && curl -L "$UP_URL" -o llama-upstream.tar.gz \ + && tar -xzf llama-upstream.tar.gz -C /app --strip-components=1 \ + && rm llama-upstream.tar.gz /tmp/upstream.json + +RUN find /app -maxdepth 1 -type f -exec chmod +x {} \; WORKDIR /app ENV PATH=/app:$PATH ENV LD_LIBRARY_PATH=/app:$LD_LIBRARY_PATH -ENV HF_HUB_ENABLE_HF_TRANSFER=1 -#ENV HF_HOME= -#ENV HUGGING_FACE_HUB_TOKEN= +ENV HF_XET_HIGH_PERFORMANCE=1 ENV LLAMA_ARG_HOST=0.0.0.0 ENV LLAMA_ARG_PORT=8090 ENV LLAMA_ARG_HF_REPO=unsloth/Qwen3.5-35B-A3B-GGUF:Q2_K_XL -#ENV LLAMA_ARG_MMPROJ_URL= -ENV LLAMA_ARG_NO_MMAP=true +ENV LLAMA_ARG_LOAD_MODE=none ENV LLAMA_ARG_CTX_SIZE=128000 +#ENV LLAMA_ARG_MMPROJ_URL= #ENV LLAMA_API_KEY="" +#ENV HF_HOME= +#ENV HUGGING_FACE_HUB_TOKEN= ENTRYPOINT ["/app/llama-server"] CMD ["--no-warmup"] diff --git a/containers/llamacpp/llamacpp-rpc.container b/containers/llamacpp/llamacpp-rpc.container index 923c92a..224ca3e 100644 --- a/containers/llamacpp/llamacpp-rpc.container +++ b/containers/llamacpp/llamacpp-rpc.container @@ -8,7 +8,7 @@ Network=internal.network # ROCm tuning AddDevice=/dev/dri/renderD128 -PodmanArgs=--group-add=keep-groups --ipc=host +PodmanArgs=--group-add=keep-groups --ipc=host --pids-limit=-1 SecurityLabelType=container_runtime_t # Cache locale dei tensori (-c), evita di ritrasferirli in rete a ogni load @@ -16,7 +16,8 @@ Volume=/srv/containers/aitools/models/rpc-cache:/root/.cache/llama.cpp/rpc # Worker RPC: espone le device locali ai server llama.cpp remoti, non e' il server principale Entrypoint=/app/ggml-rpc-server -Exec=-H 0.0.0.0 -p 50052 -c +# -t: default rpc-server e' meta' dei core host, qui il worker e' dedicato quindi usiamo tutti i core (regola in base all'host) +Exec=-H 0.0.0.0 -p 50052 -c -t 16 [Service] Restart=on-failure diff --git a/containers/llamacpp/llamacpp-vulkan.Containerfile b/containers/llamacpp/llamacpp-vulkan.Containerfile index 120c233..ed51f23 100644 --- a/containers/llamacpp/llamacpp-vulkan.Containerfile +++ b/containers/llamacpp/llamacpp-vulkan.Containerfile @@ -30,22 +30,22 @@ RUN VERSION=$(curl -s https://api.github.com/repos/ggml-org/llama.cpp/releases/l && tar -xzf llama.tar.gz -C . --strip-components=1 \ && rm llama.tar.gz -RUN chmod +x /app/llama-server +RUN find /app -maxdepth 1 -type f -exec chmod +x {} \; WORKDIR /app ENV PATH=/app:$PATH ENV LD_LIBRARY_PATH=/app:$LD_LIBRARY_PATH -ENV HF_HUB_ENABLE_HF_TRANSFER=1 -#ENV HF_HOME= -#ENV HUGGING_FACE_HUB_TOKEN= +ENV HF_XET_HIGH_PERFORMANCE=1 ENV LLAMA_ARG_HOST=0.0.0.0 ENV LLAMA_ARG_PORT=8090 ENV LLAMA_ARG_HF_REPO=unsloth/Qwen3.5-35B-A3B-GGUF:Q2_K_XL -#ENV LLAMA_ARG_MMPROJ_URL= -ENV LLAMA_ARG_NO_MMAP=true +ENV LLAMA_ARG_LOAD_MODE=none ENV LLAMA_ARG_CTX_SIZE=128000 +#ENV LLAMA_ARG_MMPROJ_URL= #ENV LLAMA_API_KEY="" +#ENV HF_HOME= +#ENV HUGGING_FACE_HUB_TOKEN= ENTRYPOINT ["/app/llama-server"] CMD ["--no-warmup"] diff --git a/containers/llamacpp/llamacpp.container b/containers/llamacpp/llamacpp.container index c3c16bf..ca486e7 100644 --- a/containers/llamacpp/llamacpp.container +++ b/containers/llamacpp/llamacpp.container @@ -9,19 +9,17 @@ Image=localhost/llamacpp:vulkan Network=internal.network PublishPort=8090:8090 -# Production - Lemonade usa Hugging Face Hub per i modelli +# Volumes Volume=/srv/containers/aitools/models/hf:/root/.cache/huggingface/hub -# Vecchia cartella -#Volume=/srv/containers/aitools/models:/root/.cache/llama.cpp # ROCm tuning AddDevice=/dev/dri/renderD128 -PodmanArgs=--group-add=keep-groups --ipc=host +PodmanArgs=--group-add=keep-groups --ipc=host --pids-limit=-1 SecurityLabelType=container_runtime_t Environment=LLAMA_ARG_HOST=0.0.0.0 Environment=LLAMA_ARG_PORT=8090 -Environment=LLAMA_ARG_NO_MMAP=true +Environment=LLAMA_ARG_LOAD_MODE=none Environment=LLAMA_ARG_CTX_SIZE=131072 Environment=LLAMA_ARG_HF_REPO=unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF:Q2_K @@ -31,6 +29,8 @@ Environment=LLAMA_ARG_SPEC_DRAFT_N_MAX=6 # RPC — worker ggml-rpc-server (llamacpp-rpc.container), lista host:port; backend insicuro, solo su internal.network #Environment=LLAMA_ARG_RPC=llamacpp-rpc:50052 +# Con RPC attivo, ripartisce i layer tra GPU locale e worker remoto (es. "1,1" = meta' e meta'); senza, tutto in locale +#Environment=LLAMA_ARG_TENSOR_SPLIT=1,1 # HF Environment=HF_HOME=/root/.cache/huggingface diff --git a/containers/vllm/check-model.py b/containers/vllm/check-model.py new file mode 100755 index 0000000..f2296f1 --- /dev/null +++ b/containers/vllm/check-model.py @@ -0,0 +1,353 @@ +#!/usr/bin/env python3 +""" +Controlla se un repo Hugging Face e' plausibilmente compatibile con questo +cluster vLLM/ROCm PRIMA di scaricarlo: dimensione totale, architettura +supportata da vLLM, formato/quantizzazione e rischi noti. + +Uso: + ./check-model.py [--node-ram-gb 128] [--gpu-util 0.75] + [--kv-reserve-gb 15] [--cluster] [--worker-ram-gb 96] + +Nota: e' un controllo euristico basato sui metadati pubblici del repo, +non sostituisce un test reale (es. compatibilita' dei kernel come Marlin +per AWQ va comunque verificata avviando vllm serve). +""" +import argparse +import json +import sys +import urllib.request + +HF_API = "https://huggingface.co/api/models/{repo_id}" +VLLM_REGISTRY_URL = ( + "https://raw.githubusercontent.com/vllm-project/vllm/main/" + "vllm/model_executor/models/registry.py" +) + + +def fetch_json(url: str): + req = urllib.request.Request(url, headers={"User-Agent": "check-model.py"}) + with urllib.request.urlopen(req, timeout=30) as resp: + return json.load(resp) + + +def fetch_text(url: str) -> str: + req = urllib.request.Request(url, headers={"User-Agent": "check-model.py"}) + with urllib.request.urlopen(req, timeout=30) as resp: + return resp.read().decode("utf-8") + + +def fetch_vllm_supported_architectures() -> set[str]: + """Estrae le chiavi (nomi architettura HF) dai dict *_MODELS in registry.py.""" + text = fetch_vllm_registry_text() + archs: set[str] = set() + for line in text.splitlines(): + line = line.strip() + if line.startswith('"') and '": (' in line: + key = line.split('"')[1] + archs.add(key) + return archs + + +_REGISTRY_TEXT_CACHE: str | None = None + + +def fetch_vllm_registry_text() -> str: + global _REGISTRY_TEXT_CACHE + if _REGISTRY_TEXT_CACHE is None: + _REGISTRY_TEXT_CACHE = fetch_text(VLLM_REGISTRY_URL) + return _REGISTRY_TEXT_CACHE + + +GITHUB_TREE_URL = "https://api.github.com/repos/vllm-project/vllm/git/trees/main?recursive=1" +_GITHUB_TREE_CACHE: dict | None = None +_PARSER_SUFFIXES = ( + "_engine_tool_parser", "_tool_parser", + "_engine_reasoning_parser", "_reasoning_parser", +) + + +def fetch_github_tree() -> dict: + global _GITHUB_TREE_CACHE + if _GITHUB_TREE_CACHE is None: + _GITHUB_TREE_CACHE = fetch_json(GITHUB_TREE_URL) + return _GITHUB_TREE_CACHE + + +def parser_base_names(tree: dict, prefix: str) -> set[str]: + """Nomi base dei parser (tool/reasoning) ricavati dai file sorgente in un dato prefisso.""" + names: set[str] = set() + for entry in tree.get("tree", []): + path = entry.get("path", "") + if not path.startswith(prefix) or not path.endswith(".py"): + continue + base = path[len(prefix):-3] + if base in ("__init__", "utils", "abstract_tool_parser", "streaming", "structural_tag_registry"): + continue + for suf in _PARSER_SUFFIXES: + if base.endswith(suf): + base = base[: -len(suf)] + break + names.add(base) + return names + + +def guess_parser_candidates(names: set[str], haystack_strings: list[str]) -> list[str]: + """Euristica: confronta i nomi base dei parser con stringhe note del repo (model_type, + architetture, tag, base_model...), solo alfanumerico, no simboli.""" + def norm(s: str) -> str: + return "".join(ch for ch in s.lower() if ch.isalnum()) + + haystacks = [norm(s) for s in haystack_strings] + candidates = [] + for name in sorted(names): + n = norm(name) + if not n: + continue + if any(n in h or h.startswith(n) for h in haystacks if h): + candidates.append(name) + return sorted(candidates) + + +def find_tool_chat_templates(tree: dict, candidates: list[str]) -> list[str]: + """Cerca in examples/tool_chat_template_*.jinja un file che corrisponda a uno dei candidati.""" + found = [] + for entry in tree.get("tree", []): + path = entry.get("path", "") + if not path.startswith("examples/tool_chat_template_") or not path.endswith(".jinja"): + continue + base = path[len("examples/tool_chat_template_"):-len(".jinja")] + base_norm = "".join(ch for ch in base.lower() if ch.isalnum()) + for c in candidates: + c_norm = "".join(ch for ch in c.lower() if ch.isalnum()) + if c_norm and (c_norm in base_norm or base_norm in c_norm): + found.append(path) + break + return found + + +def human_gb(num_bytes: float) -> str: + return f"{num_bytes / 1e9:.1f} GB" + + +GREEN = "\033[32m" +YELLOW = "\033[33m" +RED = "\033[31m" +RESET = "\033[0m" + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("repo_id", help="es. cyankiwi/Qwen3.6-35B-A3B-AWQ-4bit") + ap.add_argument("--node-ram-gb", type=float, default=128.0, help="RAM del nodo per il controllo singolo nodo (default: 128, l'head)") + ap.add_argument("--worker-ram-gb", type=float, default=96.0, help="RAM del worker, nodo piu' piccolo che vincola il budget in modalita' --cluster (default: 96)") + ap.add_argument("--gpu-util", type=float, default=0.75, help="GPU_MEMORY_UTILIZATION da usare nel calcolo (default: 0.75)") + ap.add_argument("--kv-reserve-gb", type=float, default=15.0, help="GB minimi da lasciare liberi per la KV cache (default: 15)") + ap.add_argument("--cluster", action="store_true", help="Valuta per il cluster a 2 nodi (TP=2) invece che singolo nodo") + args = ap.parse_args() + + print(f"== {args.repo_id} ==\n") + + suggestions: list[tuple[str, str]] = [("MODEL_PATH", args.repo_id)] + + try: + meta = fetch_json(HF_API.format(repo_id=args.repo_id)) + except Exception as e: + print(f"ERRORE: impossibile leggere i metadati da Hugging Face: {e}") + return 1 + + if meta.get("gated"): + print("ATTENZIONE: repo gated, servira' HF_TOKEN con accesso per scaricarlo.\n") + + siblings = [s.get("rfilename", "") for s in meta.get("siblings", [])] + gguf_files = [f for f in siblings if f.lower().endswith(".gguf")] + + tags = meta.get("tags") or [] + base_models = [ + t.split(":", 1)[1] + for t in tags + if t.startswith("base_model:") and not t.startswith("base_model:quantized:") + ] + if base_models: + print("Repo originale (base_model):", ", ".join(base_models)) + if gguf_files: + print(f" -> se e' un modello multimodale, serve --tokenizer {base_models[0]} (i GGUF multimodali non includono il tokenizer originale)") + suggestions.append(("TOKENIZER", f"{base_models[0]} # solo se multimodale")) + print() + + config = meta.get("config") or {} + architectures = config.get("architectures") or [] + model_type = config.get("model_type", "") + quant_config = config.get("quantization_config") or {} + + # --- Architettura supportata da vLLM? --- + if architectures: + try: + supported = fetch_vllm_supported_architectures() + except Exception as e: + supported = set() + print(f"(impossibile verificare la lista architetture vLLM: {e})") + print("Architettura:", ", ".join(architectures)) + for arch in architectures: + if not supported: + continue + if arch in supported: + print(f" -> OK, '{arch}' e' nel registry di vLLM") + else: + print(f" -> ATTENZIONE: '{arch}' NON risulta nel registry di vLLM (potrebbe non essere supportata)") + else: + print("Architettura: non trovata nei metadati (repo solo-GGUF? controlla il README)") + + if model_type: + print("model_type:", model_type) + + is_moe_hint = "moe" in model_type.lower() or any("moe" in a.lower() for a in architectures) + print("MoE (euristico):", "si'" if is_moe_hint else "probabilmente no") + print() + + # --- Suggerimenti tool-call-parser / reasoning-parser / chat-template --- + # Usa tutto cio' che puo' indicare la famiglia del modello: utile anche per i repo + # solo-GGUF che non espongono model_type/architectures (es. tag "minimax_m2" o + # base_model "MiniMaxAI/MiniMax-M2.7"). + haystack_strings = [model_type, *architectures, *tags, *base_models] + if any(haystack_strings): + try: + gh_tree = fetch_github_tree() + tool_names = parser_base_names(gh_tree, "vllm/tool_parsers/") + reasoning_names = parser_base_names(gh_tree, "vllm/reasoning/") + tool_candidates = guess_parser_candidates(tool_names, haystack_strings) + reasoning_candidates = guess_parser_candidates(reasoning_names, haystack_strings) + if tool_candidates: + print(f"Possibili --tool-call-parser: {', '.join(tool_candidates)} (verifica avviando vllm serve)") + suggestions.append(("ENABLE_AUTO_TOOL_CHOICE", "1")) + suggestions.append(("TOOL_CALL_PARSER", " o ".join(tool_candidates) + " # verifica")) + if reasoning_candidates: + print(f"Possibili --reasoning-parser: {', '.join(reasoning_candidates)} (verifica avviando vllm serve)") + suggestions.append(("REASONING_PARSER", " o ".join(reasoning_candidates) + " # verifica")) + templates = find_tool_chat_templates(gh_tree, tool_candidates + reasoning_candidates) + if templates: + print(f"Possibile chat template dedicato nel repo vLLM (NON incluso nella nostra immagine): {', '.join(templates)}") + suggestions.append(("CHAT_TEMPLATE", " o ".join(templates) + " # non incluso, da scaricare")) + if not tool_candidates and not reasoning_candidates: + print("Nessun tool/reasoning parser candidato trovato per euristica (nome architettura troppo diverso dai parser noti)") + except Exception as e: + print(f"(impossibile suggerire tool/reasoning parser: {e})") + print() + + # --- Quantizzazione --- + if quant_config: + method = quant_config.get("quant_method", "?") + bits = quant_config.get("bits", "?") + group_size = quant_config.get("group_size") + print(f"Quantizzazione: {method}, {bits} bit, group_size={group_size}") + if method in ("awq", "gptq") and is_moe_hint: + print(f" -> {YELLOW}CERTO{RESET}: il kernel Marlin veloce per MoE e' disabilitato in modo incondizionato su ROCm") + print(f" (vllm/.../marlin_utils.py: 'if current_platform.is_rocm(): return False', a prescindere da group_size/shape)") + print(" Fallback garantito a kernel WNA16 (piu' lento) per TUTTI i layer MoE, qualunque sia il group_size.") + elif gguf_files: + print("Quantizzazione: repo GGUF, vedi varianti sotto (nome file di solito indica il quant, es. Q4_K_M)") + else: + print("Quantizzazione: nessuna rilevata nei metadati (probabilmente pesi non quantizzati / fp16-bf16)") + print() + + # --- Dimensione --- + used_storage = meta.get("usedStorage") + safetensors_total_params = (meta.get("safetensors") or {}).get("total") + + if args.cluster: + worker_budget = args.worker_ram_gb * args.gpu_util - args.kv_reserve_gb + budget_gb = worker_budget * 2 # split simmetrico TP=2, vincolato dal nodo piu' piccolo + budget_desc = f"cluster (TP=2, vincolato dal worker da {args.worker_ram_gb:.0f}GB)" + else: + budget_gb = args.node_ram_gb * args.gpu_util - args.kv_reserve_gb + budget_desc = f"singolo nodo ({args.node_ram_gb:.0f}GB, util={args.gpu_util})" + + def rank(total_gb: float) -> int: + """0 = verde/OK, 1 = giallo/a malapena, 2 = rosso/non ci sta.""" + margin = budget_gb - total_gb + if margin < 0: + return 2 + if margin < args.kv_reserve_gb: + return 1 + return 0 + + def verdict(total_gb: float) -> str: + margin = budget_gb - total_gb + if margin < 0: + return f"{RED}NON CI STA (supera il budget di {-margin:.0f} GB){RESET}" + if margin < args.kv_reserve_gb: + return f"{YELLOW}CI STA A MALAPENA (solo {margin:.0f} GB di margine oltre la riserva KV cache){RESET}" + return f"{GREEN}OK ({margin:.0f} GB di margine oltre la riserva KV cache){RESET}" + + def status_badge(r: int) -> str: + if r == 0: + return f"{GREEN}OK{RESET}" + if r == 1: + return f"{YELLOW}A MALAPENA{RESET}" + return f"{RED}NO{RESET}" + + def margin_str(total_gb: float) -> str: + margin = budget_gb - total_gb + sign = "+" if margin >= 0 else "" + return f"{sign}{margin:.0f} GB" + + print(f"Budget {budget_desc}: ~{budget_gb:.0f} GB\n") + + if gguf_files: + try: + tree = fetch_json(HF_API.format(repo_id=args.repo_id) + "/tree/main?recursive=true") + sizes = {t["path"]: t.get("size", 0) for t in tree if t.get("type") == "file"} + except Exception: + sizes = {} + # Raggruppa per "famiglia" di quant (rimuove i suffissi -00001-of-0000N) + import re + groups: dict[str, int] = {} + for f in gguf_files: + key = re.sub(r"-\d{5}-of-\d{5}", "", f) + groups[key] = groups.get(key, 0) + sizes.get(f, 0) + known = [(key, total) for key, total in groups.items() if total] + unknown = [key for key, total in groups.items() if not total] + # Verde (piu' piccolo -> piu' grande), separatore, giallo, separatore, rosso. + known.sort(key=lambda kt: (rank(kt[1] / 1e9), kt[1])) + + all_names = [k for k, _ in known] + unknown + name_w = max([len("VARIANTE")] + [len(n) for n in all_names]) + size_w = max([len("DIMENSIONE")] + [len(human_gb(t)) for _, t in known]) + margin_w = max([len("MARGINE")] + [len(margin_str(t / 1e9)) for _, t in known]) + + print("File GGUF trovati:") + print(f" {'VARIANTE':<{name_w}} {'DIMENSIONE':>{size_w}} {'MARGINE':>{margin_w}} STATO") + print(f" {'-' * name_w} {'-' * size_w} {'-' * margin_w} -----") + last_rank = None + for key, total in known: + r = rank(total / 1e9) + if last_rank is not None and r != last_rank: + print(f" {'-' * name_w} {'-' * size_w} {'-' * margin_w} -----") + print(f" {key:<{name_w}} {human_gb(total):>{size_w}} {margin_str(total / 1e9):>{margin_w}} {status_badge(r)}") + last_rank = r + for key in sorted(unknown): + print(f" {key:<{name_w}} {'?':>{size_w}} {'?':>{margin_w}} ?") + else: + total_bytes = used_storage + if total_bytes: + print(f"Dimensione totale repo (usedStorage): {human_gb(total_bytes)}") + if safetensors_total_params: + print(f"Parametri totali (safetensors): {safetensors_total_params / 1e9:.1f} B") + if total_bytes: + print(f"-> {verdict(total_bytes / 1e9)}") + else: + print("Dimensione non determinabile automaticamente.") + + print() + print("Variabili quadlet suggerite:") + var_w = max(len("VARIABILE"), max(len(v) for v, _ in suggestions)) + val_w = max(len("VALORE"), max(len(v) for _, v in suggestions)) + print(f" {'VARIABILE':<{var_w}} VALORE") + print(f" {'-' * var_w} {'-' * min(val_w, 60)}") + for name, value in suggestions: + print(f" {name:<{var_w}} {value}") + + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/containers/vllm/entrypoint.sh b/containers/vllm/entrypoint.sh index 129b98c..a1ab159 100755 --- a/containers/vllm/entrypoint.sh +++ b/containers/vllm/entrypoint.sh @@ -10,6 +10,8 @@ RAY_PORT="${RAY_PORT:-6379}" NUM_GPUS="${NUM_GPUS:-1}" if [ "$CLUSTER" -eq 0 ]; then + # Nessun cluster: non deve dipendere dalla rete Thunderbolt usata per Ray. + unset VLLM_HOST_IP NCCL_SOCKET_IFNAME GLOO_SOCKET_IFNAME : "${MODEL_PATH:?MODEL_PATH non impostato}" # TOKENIZER e TRUST_REMOTE_CODE sono opzionali: se non impostate, vllm @@ -32,6 +34,12 @@ if [ "$CLUSTER" -eq 0 ]; then if [ -n "${REASONING_PARSER:-}" ]; then EXTRA_ARGS+=(--reasoning-parser "${REASONING_PARSER}") fi + if [ -n "${ATTENTION_BACKEND:-}" ]; then + EXTRA_ARGS+=(--attention-backend "${ATTENTION_BACKEND}") + fi + if [ -n "${CHAT_TEMPLATE:-}" ]; then + EXTRA_ARGS+=(--chat-template "${CHAT_TEMPLATE}") + fi echo "[entrypoint] Modalita' singolo nodo (no cluster): avvio vllm serve senza Ray" exec vllm serve "${MODEL_PATH}" \ @@ -73,6 +81,12 @@ elif [ "$CLUSTER" -eq 1 ]; then if [ -n "${REASONING_PARSER:-}" ]; then EXTRA_ARGS+=(--reasoning-parser "${REASONING_PARSER}") fi + if [ -n "${ATTENTION_BACKEND:-}" ]; then + EXTRA_ARGS+=(--attention-backend "${ATTENTION_BACKEND}") + fi + if [ -n "${CHAT_TEMPLATE:-}" ]; then + EXTRA_ARGS+=(--chat-template "${CHAT_TEMPLATE}") + fi echo "[entrypoint] Avvio vllm serve" exec vllm serve "${MODEL_PATH}" \ diff --git a/containers/vllm/vllm.container b/containers/vllm/vllm.container index 12d578e..693f183 100644 --- a/containers/vllm/vllm.container +++ b/containers/vllm/vllm.container @@ -40,7 +40,11 @@ Environment=RAY_PORT=6379 Environment=NUM_GPUS=1 # vLLM serve (usati solo dal nodo head, CLUSTER=1) +# MODEL_PATH puo' essere un path locale a un file GGUF (come sotto) oppure +# direttamente un repo id Hugging Face (es. cyankiwi/Qwen3.6-35B-A3B-AWQ-4bit): +# in quel caso vLLM lo scarica da solo nella cache indicata da HF_HOME. Environment=MODEL_PATH=/root/.cache/huggingface/hub/models--unsloth--MiniMax-M2.7-GGUF/snapshots/d2a05ccf69491b03db0cc40b335aec14bdaf7198/UD-IQ4_NL/MiniMax-M2.7-UD-IQ4_NL-00001-of-00004.gguf +#Environment=MODEL_PATH=cyankiwi/Qwen3.6-35B-A3B-AWQ-4bit #Environment=TOKENIZER=unsloth/MiniMax-M2.7-GGUF Environment=SERVED_MODEL_NAME=MiniMax-M2.7 Environment=TRUST_REMOTE_CODE=0 @@ -52,14 +56,15 @@ Environment=TENSOR_PARALLEL_SIZE=2 Environment=DTYPE=float16 Environment=KV_CACHE_DTYPE=fp8 Environment=DISTRIBUTED_TIMEOUT_SECONDS=1800 -# Default vLLM sceglie ROCM_ATTN; scommenta per forzare il backend Triton se persistono i crash con ROCM_ATTN. -#Environment=VLLM_ATTENTION_BACKEND=TRITON_ATTN +# Default vLLM sceglie ROCM_ATTN. +#Environment=ATTENTION_BACKEND=TRITON_ATTN # Tool/function calling (richiesto dai framework di agenti per il tool_choice="auto"). # Parser registrato in vLLM per MiniMax M2: "minimax_m2". Environment=ENABLE_AUTO_TOOL_CHOICE=1 Environment=TOOL_CALL_PARSER=minimax_m2 Environment=REASONING_PARSER=minimax_m2 +#Environment=CHAT_TEMPLATE=/opt/vllm-rocm/chat-templates/tool_chat_template_gemma4.jinja # Riduce la frammentazione di memoria HIP Environment=PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True