apiVersion: apps.nvidia.com/v1alpha1 kind: NIMCache metadata: name: qwen3-32b-instruct namespace: nim-service spec: source: ngc: modelPuller: nvcr.io/nim/qwen/qwen3-32b-dgx-spark:1.1.0-variant pullSecret: ngc-secret authSecret: ngc-api-secret model: engine: "vllm" tensorParallelism: "1" storage: pvc: create: true size: "100Gi" volumeAccessMode: ReadWriteOnce ---