"""Isoler la cause du charabia vLLM sous CC ON (H100 TDX).
  python vllm_cc_test2.py nopin | v1runner | sync
nopin    : retire pin_memory=True partout (torch.empty/zeros/tensor...), sans multiprocessing
v1runner : VLLM_USE_V2_MODEL_RUNNER=0 (ancien model runner), eager
sync     : CUDA_LAUNCH_BLOCKING=1, eager
"""
import json, os, sys, time
mode = sys.argv[1]
os.environ["VLLM_ENABLE_V1_MULTIPROCESSING"] = "0"
if mode == "v1runner":
    os.environ["VLLM_USE_V2_MODEL_RUNNER"] = "0"
if mode == "sync":
    os.environ["CUDA_LAUNCH_BLOCKING"] = "1"
import torch
patched = 0
if mode == "nopin":
    def wrap(fn):
        def inner(*a, **k):
            global patched
            if k.get("pin_memory"):
                k["pin_memory"] = False
                patched += 1
            return fn(*a, **k)
        return inner
    for name in ("empty", "zeros", "ones", "tensor", "full", "arange", "empty_strided", "as_tensor"):
        if hasattr(torch, name):
            setattr(torch, name, wrap(getattr(torch, name)))
    torch.Tensor.pin_memory = lambda self, *a, **k: self
MODEL = "Qwen/Qwen2.5-1.5B-Instruct"
QUESTIONS = [
    "What is the capital of France? Answer in one word.",
    "What is 17 + 25? Answer with the number only.",
    "Name three primary colors, separated by commas.",
    "Translate 'good morning' into French.",
    "In one sentence, what does a CPU do?",
    "Write the word 'banana' backwards.",
]
t0 = time.time()
from vllm import LLM, SamplingParams
import vllm
llm = LLM(model=MODEL, enforce_eager=True, gpu_memory_utilization=0.6, max_model_len=2048)
res = llm.chat([[{"role": "user", "content": q}] for q in QUESTIONS], SamplingParams(temperature=0, max_tokens=48))
outs = [r.outputs[0].text.strip() for r in res]
print(json.dumps({"mode": mode, "secondes": round(time.time() - t0, 1), "pin_patches": patched, "vllm": vllm.__version__, "reponses": outs}, ensure_ascii=False))
