from chutes.chute import NodeSelector from chutes.chute.template.vllm import build_vllm_chute chute = build_vllm_chute( tee=True, username="chutes", model_name="Qwen/Qwen3-235B-A22B-Thinking-2507", revision="4cd68849b2a0fadb84866c703b133aa8c8636130", image="chutes/vllm:nightly-2026030502", concurrency=40, node_selector=NodeSelector( gpu_count=8, include=["h200", "b200", "b300"], ), engine_args=( "--enable-auto-tool-choice " "--tool-call-parser hermes " "--reasoning-parser deepseek_r1 " "--max-num-batched-tokens 4096 " "--max-completion-tokens 16384 " "--max-stream-completion-tokens 65536 " "--max-num-seqs 40 " "--max-cudagraph-capture-size 40 " "--gpu-memory-utilization 0.85" ), ) chute.chute._name = "Qwen/Qwen3-235B-A22B-Thinking-2507-TEE"