# DFLASH2 tony-lane image — built directly on the proven day-0 stack that
# tonyd2wild served 46.9 tok/s (C1 warm) / 0.741 acceptance with on 2x DGX
# Spark, fp8 KV, DFLASH2. Base = their published sm121-v8; overlay = the
# vendored overlay-dflash2/ (their exact patch set: registry+select,
# GLM aux capture, drafter KV group). No Ray, no Mia kernel layer.
#
# Build context = repo ROOT so `files/overlay-dflash2` is reachable:
#   docker build --network=host -t glm53-dflash2-tony:v1 -f tony-lane/Dockerfile .
# (accepts arm64 on the Spark; do NOT add --platform.)

FROM radixark/vllm-glm53-flash:sm121-v8

# Optional InstantTensor direct-I/O loader (v9's trick, 15x faster loads).
# CONFIRMED WORKING on this kit (2026-08-29: boot + serve). tonyd2wild's md
# reports 4/4 silent TP2 deaths ~1 min after load on their stack at every KV
# budget — the same-layer nccl re-pin to 2.30.7 here may be why we survive
# (unproven). Off by default only because safetensors remains the proven
# default: build with `--build-arg INSTANTTENSOR=1` and launch with
# `LOAD_FORMAT=instanttensor`. Same-layer nccl re-pin is REQUIRED
# (instanttensor downgrades nccl->2.29.7, fabric-fatal).
ARG INSTANTTENSOR=0
RUN if [ "$INSTANTTENSOR" = "1" ]; then \
        pip install -q instanttensor \
     && pip install -q nvidia-nccl-cu13==2.30.7 \
     && pip show nvidia-nccl-cu13 | grep -q "Version: 2.30.7" \
     && python3 -c "import instanttensor" \
     && echo "[glm53-tony] instanttensor ready (nccl re-pinned 2.30.7)"; \
    fi

COPY files/overlay-dflash2/ /opt/dflash2/

RUN cp /opt/dflash2/qwen3_dflash2.py \
        /usr/local/lib/python3.12/dist-packages/vllm/model_executor/models/ \
 && mkdir -p /usr/local/lib/python3.12/dist-packages/vllm/v1/worker/gpu/spec_decode/dflash2 \
 && cp /opt/dflash2/dflash2/__init__.py \
        /opt/dflash2/dflash2/speculator.py \
        /usr/local/lib/python3.12/dist-packages/vllm/v1/worker/gpu/spec_decode/dflash2/ \
 && python3 /opt/dflash2/patch_registry_and_select.py \
        /usr/local/lib/python3.12/dist-packages/vllm \
 && python3 /opt/dflash2/patch_glm_aux_capture.py \
 && python3 /opt/dflash2/patch_glm5_drafter_group.py \
 && python3 -c "from vllm.model_executor.models.registry import ModelRegistry; \
assert 'DFlash2DraftModel' in ModelRegistry.get_supported_archs(); \
print('[glm53-tony] registry: DFlash2 OK')" \
 && python3 -c "import vllm.model_executor.models.qwen3_dflash2, \
vllm.v1.worker.gpu.spec_decode.dflash2.speculator; \
print('[glm53-tony] dflash2 modules import OK')"
