diff --git a/common/configure_pip_index.sh b/common/configure_pip_index.sh new file mode 100644 index 0000000..6745aaa --- /dev/null +++ b/common/configure_pip_index.sh @@ -0,0 +1,27 @@ +#!/bin/sh + +configure_pip_index() { + secret_file="${PIP_INDEX_URL_SECRET_FILE:-/run/secrets/pip_index_url}" + + if [ -r "${secret_file}" ]; then + private_index_url=$(cat "${secret_file}") + case "${private_index_url}" in + https://*) ;; + *) + echo "Error: private pip index secret must contain an HTTPS URL" >&2 + return 1 + ;; + esac + + export PIP_CONFIG_FILE=/dev/null + export PIP_INDEX_URL="${private_index_url}" + unset PIP_EXTRA_INDEX_URL + echo "Using the configured private pip index" + return 0 + fi + + if [ "${REQUIRE_PIP_INDEX_SECRET:-false}" = "true" ]; then + echo "Error: required private pip index secret is unavailable" >&2 + return 1 + fi +} diff --git a/common/sagemaker_args.py b/common/sagemaker_args.py new file mode 100644 index 0000000..ff9416f --- /dev/null +++ b/common/sagemaker_args.py @@ -0,0 +1,199 @@ +"""Translate SM_VLLM_* environment variables into vLLM server CLI arguments. + +SageMaker passes configuration as environment variables, so the entrypoint has to +turn ``SM_VLLM_TENSOR_PARALLEL_SIZE=8`` into ``--tensor-parallel-size 8``. +The translation mirrors what vLLM already does for ``--config file.yaml`` in +``vllm/utils/argparse_utils.py``: + + * JSON array -> the flag plus one argv token per element + * JSON object -> the flag plus a single argv token (the object) + * anything else -> the flag plus the raw value as a single token + +Because the decision is made from the *value's* syntax rather than a per-flag table, +no list of flags has to be maintained as vLLM adds arguments. The exception is the +handful of flags where vLLM deliberately wants an array as one token; those are +listed in SINGLE_TOKEN_FLAGS below. + +Tokens are written to stdout NUL-delimited so the shell can read them back into an +array without word-splitting values that contain spaces or newlines. Informational +messages go to stderr to keep stdout parseable. +""" + +import json +import os +import sys +from typing import Any, Iterable, List, Mapping, Optional + +PREFIX = "SM_VLLM_" +ARG_PREFIX = "--" +DEFAULT_PORT = "8080" +MODEL_DIR = "/opt/ml/model" + +# Flags whose value stays a single argv token even when it is a JSON array. +# vLLM strips nargs from these on purpose (see FrontendArgs._customize_cli_kwargs in +# vllm/entrypoints/openai/cli_args.py): the first three are typed ``json.loads`` so the +# array *is* the value, and --middleware is ``action="append"``, taking one value per +# occurrence. +SINGLE_TOKEN_FLAGS = frozenset( + { + "--allowed-origins", + "--allowed-methods", + "--allowed-headers", + "--middleware", + } +) + + +def flag_for(env_key: str) -> str: + """``SM_VLLM_TENSOR_PARALLEL_SIZE`` -> ``--tensor-parallel-size``.""" + name = env_key[len(PREFIX) :].lower().replace("_", "-") + return f"{ARG_PREFIX}{name}" + + +def as_token(element: Any) -> str: + """Render one element of a JSON array as a single argv token.""" + if isinstance(element, (dict, list)): + return json.dumps(element, separators=(",", ":")) + if isinstance(element, bool): + return "true" if element else "false" + if element is None: + return "" + return str(element) + + +def json_object_sequence(value: str) -> Optional[List[dict]]: + """Parse ``{...} {...}`` into its objects. + + Returns None if the string is not a whitespace-separated run of JSON objects, so + callers can fall back to passing the value through untouched. + """ + decoder = json.JSONDecoder() + objects: List[dict] = [] + index = 0 + length = len(value) + while index < length: + try: + parsed, end = decoder.raw_decode(value, index) + except ValueError: + return None + if not isinstance(parsed, dict): + return None + objects.append(parsed) + index = end + while index < length and value[index].isspace(): + index += 1 + return objects or None + + +def tokens_for(flag: str, value: str) -> Optional[List[str]]: + """Return the argv tokens that follow `flag`. + + None means "omit the flag entirely". That covers both an empty value (e.g. a + templated env var defaulted to ``""``) and an empty JSON array, since emitting a + bare value-taking flag or a nargs='+' flag with zero values makes argparse reject + the arguments and the server fail to start. + """ + if not value.strip(): + return None + + stripped = value.strip() + + if flag in SINGLE_TOKEN_FLAGS: + return [value] + + if stripped.startswith("[") and stripped.endswith("]"): + try: + parsed = json.loads(stripped) + except ValueError: + return [value] + if isinstance(parsed, list): + if not parsed: + return None + return [as_token(element) for element in parsed] + return [value] + + if stripped.startswith("{") and stripped.endswith("}"): + objects = json_object_sequence(stripped) + if objects is not None and len(objects) > 1: + return [json.dumps(obj, separators=(",", ":")) for obj in objects] + # A single JSON object is passed through verbatim + return [value] + + return [value] + + +def resolve_model(env: Mapping[str, str], model_dir: str) -> List[str]: + """Pick the model source when SM_VLLM_MODEL is not set. + + Precedence: an explicit SM_VLLM_MODEL (handled by the generic loop, so nothing is + added here), then a populated model dir, then HF_MODEL_ID. + """ + if env.get(f"{PREFIX}MODEL"): + return [] + + if os.path.isdir(model_dir) and os.listdir(model_dir): + log(f"INFO: {PREFIX}MODEL not set, auto-detected model at {model_dir}") + return ["--model", model_dir] + + if env.get("HF_MODEL_ID"): + log(f"INFO: {PREFIX}MODEL not set, using HF_MODEL_ID={env['HF_MODEL_ID']}") + return ["--model", env["HF_MODEL_ID"]] + + log( + f"WARNING: No model specified. Set {PREFIX}MODEL, HF_MODEL_ID, " + f"or mount a model to {model_dir}." + ) + return [] + + +def log(message: str) -> None: + print(message, file=sys.stderr) + + +def build_args(env: Mapping[str, str], model_dir: str = MODEL_DIR) -> List[str]: + """Build the full argv list (minus the program itself) for the vLLM server.""" + # Only supply the default port when the user hasn't set one. Otherwise the loop + # below also emits SM_VLLM_PORT as --port, producing a duplicate flag that relies + # on argparse silently keeping the last occurrence. + args: List[str] = [] + if not env.get(f"{PREFIX}PORT"): + args += ["--port", DEFAULT_PORT] + args += resolve_model(env, model_dir) + + for key in sorted(env): + if not key.startswith(PREFIX): + continue + flag = flag_for(key) + value = env[key] + lowered = value.strip().lower() + + # Boolean flags: true -> bare flag, false -> omitted entirely. + if lowered == "true": + args.append(flag) + continue + if lowered == "false": + continue + + tokens = tokens_for(flag, value) + if tokens is None: + log(f"WARNING: {key} is empty; skipping {flag}.") + continue + args.append(flag) + args += tokens + + return args + + +def emit(tokens: Iterable[str]) -> None: + """Write tokens NUL-delimited for `mapfile -t -d ''` on the shell side.""" + sys.stdout.write("".join(f"{token}\0" for token in tokens)) + + +def main() -> None: + args = build_args(os.environ) + log(f"INFO: vLLM server arguments: {args}") + emit(args) + + +if __name__ == "__main__": + main() diff --git a/common/serve.sh b/common/serve.sh new file mode 100644 index 0000000..5ecc38e --- /dev/null +++ b/common/serve.sh @@ -0,0 +1,26 @@ +#!/bin/bash +# SageMaker hosting entrypoint for the vLLM Neuron DLC. +# +# SageMaker launches an inference container as `docker run serve`. This image's +# ENTRYPOINT (common/vllm_entrypoint.py) execs argv as-is, so exposing an executable +# named `serve` on PATH satisfies SageMaker without changing the ENTRYPOINT, keeping the +# existing EC2/k8s usage (`docker run vllm serve ...`) working unchanged. +# +# Adapted from aws/deep-learning-containers scripts/docker/vllm/sagemaker_entrypoint.sh +# (Apache-2.0). vLLM >= 0.24 already registers /ping and /invocations via +# vllm.entrypoints.serve.sagemaker.api_router, so no routing middleware is needed here. +set -euo pipefail + +ARGS_FILE=$(mktemp) +trap 'rm -f "${ARGS_FILE}"' EXIT +if ! python3 /usr/local/bin/sagemaker_args.py >"${ARGS_FILE}"; then + echo "ERROR: failed to build vLLM arguments from SM_VLLM_* environment variables" >&2 + exit 1 +fi + +ARGS=() +while IFS= read -r -d '' token; do + ARGS+=("${token}") +done <"${ARGS_FILE}" + +exec standard-supervisor python3 -m vllm.entrypoints.openai.api_server "${ARGS[@]}" diff --git a/vllm-omni/inference/0.24.0.1.1.0/Dockerfile.neuronx b/vllm-omni/inference/0.24.0.1.1.0/Dockerfile.neuronx new file mode 100644 index 0000000..88cb8a6 --- /dev/null +++ b/vllm-omni/inference/0.24.0.1.1.0/Dockerfile.neuronx @@ -0,0 +1,311 @@ +ARG BUILD_STAGE=prod + +FROM public.ecr.aws/docker/library/ubuntu:24.04 AS base + +LABEL dlc_major_version="1" +LABEL maintainer="Amazon AI" + +ARG DEBIAN_FRONTEND=noninteractive +ARG PIP=pip3 +ARG PYTHON=python3.13 +ARG PYTHON_VERSION=3.13.13 +ARG TORCHSERVE_VERSION=0.11.0 +ARG PYPI_SIMPLE_URL="https://pypi.org/simple/" + + +# See http://bugs.python.org/issue19846 +ENV LANG=C.UTF-8 +ENV LD_LIBRARY_PATH="/opt/aws/neuron/lib" +ENV LD_LIBRARY_PATH="${LD_LIBRARY_PATH}:/opt/amazon/efa/lib" +ENV LD_LIBRARY_PATH="${LD_LIBRARY_PATH}:/opt/amazon/efa/lib64" +ENV LD_LIBRARY_PATH="${LD_LIBRARY_PATH}:/lib/x86_64-linux-gnu" +ENV LD_LIBRARY_PATH="${LD_LIBRARY_PATH}:/opt/conda/lib/" +ENV PATH=/opt/conda/bin:/opt/aws/neuron/bin:$PATH +ENV PATH="${PATH}:/opt/amazon/efa/bin" + +RUN apt-get update \ + && apt-get upgrade -y \ + && apt-get install -y --no-install-recommends \ + apt-transport-https \ + build-essential \ + ca-certificates \ + cmake \ + curl \ + ffmpeg \ + gcc \ + git \ + gnupg2 \ + gpg-agent \ + jq \ + libgl1 \ + libgl1-mesa-dri \ + libglib2.0-0 \ + libsm6 \ + libxext6 \ + libxrender-dev \ + libcap-dev \ + libhwloc-dev \ + openssh-client \ + openjdk-11-jdk \ + unzip \ + wget \ + zlib1g-dev \ + && rm -rf /var/lib/apt/lists/* \ + && rm -rf /tmp/tmp* \ + && apt-get clean + + +# https://github.com/docker-library/openjdk/issues/261 https://github.com/docker-library/openjdk/pull/263/files +RUN keytool -importkeystore -srckeystore /etc/ssl/certs/java/cacerts -destkeystore /etc/ssl/certs/java/cacerts.jks -deststoretype JKS -srcstorepass changeit -deststorepass changeit -noprompt; \ + mv /etc/ssl/certs/java/cacerts.jks /etc/ssl/certs/java/cacerts; \ + /var/lib/dpkg/info/ca-certificates-java.postinst configure; + +RUN curl -L -o ~/miniforge.sh https://github.com/conda-forge/miniforge/releases/latest/download/Miniforge3-Linux-x86_64.sh \ + && chmod +x ~/miniforge.sh \ + && ~/miniforge.sh -b -p /opt/conda \ + && rm ~/miniforge.sh \ + && /opt/conda/bin/mamba install -c conda-forge -y \ + python=$PYTHON_VERSION \ + pyopenssl \ + cython \ + mkl-include \ + mkl \ + parso \ + # Below 2 are included in miniconda base, but not mamba so need to install + conda-content-trust \ + charset-normalizer \ + && /opt/conda/bin/conda clean -ya + +RUN /opt/conda/bin/mamba install -c conda-forge \ + python=$PYTHON_VERSION \ + scikit-learn \ + h5py \ + requests \ + && conda clean -ya \ + && pip install --upgrade pip \ + --trusted-host pypi.org --trusted-host files.pythonhosted.org \ + && ln -s /opt/conda/bin/pip /usr/local/bin/pip3 \ + && pip install \ + enum-compat \ + ipython \ + && rm -rf ~/.cache/pip/* + +# Install EFA +RUN apt-get update \ + && cd $HOME \ + && curl -O https://efa-installer.amazonaws.com/aws-efa-installer-latest.tar.gz \ + && wget https://efa-installer.amazonaws.com/aws-efa-installer.key && gpg --import aws-efa-installer.key \ + && cat aws-efa-installer.key | gpg --fingerprint \ + && wget https://efa-installer.amazonaws.com/aws-efa-installer-latest.tar.gz.sig && gpg --verify ./aws-efa-installer-latest.tar.gz.sig \ + && tar -xf aws-efa-installer-latest.tar.gz \ + && cd aws-efa-installer \ + && ./efa_installer.sh -y -g --skip-kmod --skip-limit-conf --no-verify \ + && cd $HOME \ + && rm -rf /var/lib/apt/lists/* \ + && rm -rf /tmp/tmp* \ + && apt-get clean + +# Stub libcuda.so.1 so the NIXL LIBFABRIC backend can dlopen on Neuron (no CUDA driver present). +# The plugin references three CUDA *driver* entrypoints only for an optional GPUDirect path that +# is never exercised on Neuron; defining them as no-ops satisfies RTLD_NOW without a real driver. +RUN printf '%s\n' \ + 'int cuCtxSetCurrent(void* c){return 999;}' \ + 'int cuDeviceGetPCIBusId(char* s, int n, int d){return 999;}' \ + 'int cuPointerGetAttributes(void* a, void* b, unsigned long long p){return 999;}' \ + > /tmp/cuda_stub.c \ + && gcc -shared -fPIC -Wl,-soname,libcuda.so.1 -o /usr/local/lib/libcuda.so.1 /tmp/cuda_stub.c \ + && rm -f /tmp/cuda_stub.c \ + && ldconfig + +COPY --chmod=755 vllm_entrypoint.py neuron-monitor.sh deep_learning_container.py /usr/local/bin/ + +# SageMaker hosting support. SageMaker launches an inference container as +# `docker run serve`, so exposing an executable named `serve` on PATH lets the +# existing ENTRYPOINT (vllm_entrypoint.py, which execs its argv) start the vLLM server. +COPY --chmod=755 serve.sh /usr/local/bin/serve +COPY --chmod=755 sagemaker_args.py /usr/local/bin/sagemaker_args.py + +### Mount Point ### +# When launching the container, mount the code directory to /workspace +ARG APP_MOUNT=/workspace +VOLUME ${APP_MOUNT} +WORKDIR ${APP_MOUNT}/vllm + +# Install AWS CLI +RUN curl "https://awscli.amazonaws.com/awscli-exe-linux-x86_64.zip" -o "awscliv2.zip" \ + && unzip awscliv2.zip \ + && ./aws/install \ + && rm -rf awscliv2.zip aws + +RUN ${PIP} install --no-cache-dir -U \ + "opencv-python" \ + "pandas" \ + "boto3" \ + "cryptography" \ + "pytest" \ + "wheel" \ + "jinja2" \ + uv \ + torchserve==${TORCHSERVE_VERSION} \ + torch-model-archiver==${TORCHSERVE_VERSION} \ + && rm -rf ~/.cache/pip/* + +RUN useradd -m model-server \ + && mkdir -p /home/model-server/tmp /opt/ml/model \ + && chown -R model-server /home/model-server /opt/ml/model +COPY config.properties /home/model-server + +# Compliance +RUN HOME_DIR=/root \ + && curl -o ${HOME_DIR}/oss_compliance.zip https://aws-dlinfra-utilities.s3.amazonaws.com/oss_compliance.zip \ + && unzip ${HOME_DIR}/oss_compliance.zip -d ${HOME_DIR}/ \ + && cp ${HOME_DIR}/oss_compliance/test/testOSSCompliance /usr/local/bin/testOSSCompliance \ + && chmod +x /usr/local/bin/testOSSCompliance \ + && chmod +x ${HOME_DIR}/oss_compliance/generate_oss_compliance.sh \ + && ${HOME_DIR}/oss_compliance/generate_oss_compliance.sh ${HOME_DIR} ${PYTHON} \ + && rm -rf ${HOME_DIR}/oss_compliance* \ + # conda leaves an empty /root/.cache/conda/notices.cache file which is not removed by conda clean -ya + && rm -rf ${HOME_DIR}/.cache/conda + +# Setting up APT and PIP repo for neuron artifacts +ARG NEURON_APT_REPO=apt.repos.neuron.amazonaws.com +ARG NEURON_PIP_REPO=pip.repos.neuron.amazonaws.com +RUN --mount=type=secret,id=neuron_repo \ + mkdir -p /etc/apt/keyrings \ + && echo "deb [signed-by=/etc/apt/keyrings/neuron.gpg] https://${NEURON_APT_REPO} jammy main" > /etc/apt/sources.list.d/neuron.list \ + && NEURON_REPO_CRED="$([ -s /run/secrets/neuron_repo ] && cat /run/secrets/neuron_repo || true)" \ + && curl $([ -n "${NEURON_REPO_CRED}" ] && echo "-u ${NEURON_REPO_CRED}") --retry 3 --retry-delay 1 --retry-all-errors -fSL "https://${NEURON_APT_REPO}/GPG-PUB-KEY-AMAZON-AWS-NEURON.PUB" | gpg --dearmor > /etc/apt/keyrings/neuron.gpg + +# Neuron SDK components version numbers +ARG NEURONX_COLLECTIVES_LIB_VERSION=2.34.10.0-74eaafac6 +ARG NEURONX_RUNTIME_LIB_VERSION=2.34.10.0-ac18d186d +ARG NEURONX_TOOLS_VERSION=2.32.28.0-526c2b7f6 + +ARG NEURONX_CC_VERSION=2.27.5334.0+f702b353 +ARG NKI_VERSION=0.6.0+31049202112.g85070674 +ARG NEURON_AGENTIC_DEVELOPMENT_VERSION=1.3 +ARG LIBTORCH_NEURONX_LITE_VERSION=2.11.0.1.0.2651+723ba691 + +# Primary repo: vllm-omni-neuron +ARG GITHUB_REPO=https://github.com/aws-neuron/vllm-omni-neuron.git +ARG GITHUB_REPO_BRANCH=release-0.24.0.0.1.0 + +# Secondary repo: vllm-neuron +ARG GITHUB_REPO_2=https://github.com/vllm-project/vllm-neuron.git +ARG GITHUB_REPO_BRANCH_2=release-0.24.0.1.1.0 + +# Configure SSH access +RUN mkdir -p /root/.ssh \ + && echo "StrictHostKeyChecking no" >> /root/.ssh/config \ + && ssh-keyscan -t rsa github.com >> /root/.ssh/known_hosts + +# Clone vllm-omni-neuron repository +RUN --mount=type=secret,id=containers_github_ssh_key,target=/root/.ssh/id_ed25519,mode=0600 \ + git clone -b ${GITHUB_REPO_BRANCH} ${GITHUB_REPO} /opt/vllm-omni-neuron + +# Clone vllm-neuron repository +RUN --mount=type=secret,id=containers_github_ssh_key,target=/root/.ssh/id_ed25519,mode=0600 \ + git clone -b ${GITHUB_REPO_BRANCH_2} ${GITHUB_REPO_2} /opt/vllm-neuron + +FROM base AS repo + + +# Install Neuron components from the apt and pip repos (latest versions) +RUN --mount=type=secret,id=neuron_repo \ + mkdir -p /etc/apt/auth.conf.d \ + && if [ -s /run/secrets/neuron_repo ]; then \ + CRED="$(cat /run/secrets/neuron_repo)"; \ + printf 'machine %s login %s password %s\n' "${NEURON_APT_REPO}" "${CRED%%:*}" "${CRED#*:}" > /etc/apt/auth.conf.d/neuron.conf; \ + chmod 600 /etc/apt/auth.conf.d/neuron.conf; \ + fi \ + && apt-get update \ + && apt-get install -y \ + aws-neuronx-tools \ + aws-neuronx-collectives \ + aws-neuronx-runtime-lib \ + && rm -rf /var/lib/apt/lists/* \ + && rm -rf /tmp/tmp* \ + && apt-get clean \ + && rm -f /etc/apt/auth.conf.d/neuron.conf + +RUN --mount=type=secret,id=neuron_repo \ + PIP_REPO_URL="https://${NEURON_PIP_REPO}" \ + && if [ -s /run/secrets/neuron_repo ]; then \ + CRED="$(cat /run/secrets/neuron_repo)"; \ + ENC="$(printf '%s' "$CRED" | python3 -c 'import urllib.parse,sys; u,p=sys.stdin.read().split(":",1); print(urllib.parse.quote(u,safe="")+":"+urllib.parse.quote(p,safe=""))')"; \ + PIP_REPO_URL="https://${ENC}@${NEURON_PIP_REPO}"; \ + fi \ + && ${PIP} install --no-cache-dir \ + --index-url ${PIP_REPO_URL} \ + --trusted-host ${NEURON_PIP_REPO} \ + --extra-index-url ${PYPI_SIMPLE_URL} \ + "islpy==2026.1" \ + neuronx-cc \ + nki \ + neuron_agentic_development \ + libtorch-neuronx-lite \ + /opt/vllm-omni-neuron \ + /opt/vllm-neuron \ + && rm -rf ~/.cache/pip/* + +FROM base AS prod + +# Install Neuron components with specific versions +RUN --mount=type=secret,id=neuron_repo \ + mkdir -p /etc/apt/auth.conf.d \ + && if [ -s /run/secrets/neuron_repo ]; then \ + CRED="$(cat /run/secrets/neuron_repo)"; \ + printf 'machine %s login %s password %s\n' "${NEURON_APT_REPO}" "${CRED%%:*}" "${CRED#*:}" > /etc/apt/auth.conf.d/neuron.conf; \ + chmod 600 /etc/apt/auth.conf.d/neuron.conf; \ + fi \ + && apt-get update \ + && apt-get install -y \ + aws-neuronx-tools=$NEURONX_TOOLS_VERSION \ + aws-neuronx-collectives=$NEURONX_COLLECTIVES_LIB_VERSION \ + aws-neuronx-runtime-lib=$NEURONX_RUNTIME_LIB_VERSION \ + && rm -rf /var/lib/apt/lists/* \ + && rm -rf /tmp/tmp* \ + && apt-get clean \ + && rm -f /etc/apt/auth.conf.d/neuron.conf + +RUN --mount=type=secret,id=neuron_repo \ + PIP_REPO_URL="https://${NEURON_PIP_REPO}" \ + && if [ -s /run/secrets/neuron_repo ]; then \ + CRED="$(cat /run/secrets/neuron_repo)"; \ + ENC="$(printf '%s' "$CRED" | python3 -c 'import urllib.parse,sys; u,p=sys.stdin.read().split(":",1); print(urllib.parse.quote(u,safe="")+":"+urllib.parse.quote(p,safe=""))')"; \ + PIP_REPO_URL="https://${ENC}@${NEURON_PIP_REPO}"; \ + fi \ + && ${PIP} install --no-cache-dir \ + --index-url ${PIP_REPO_URL} \ + --trusted-host ${NEURON_PIP_REPO} \ + --extra-index-url ${PYPI_SIMPLE_URL} \ + "islpy==2026.1" \ + neuronx-cc==$NEURONX_CC_VERSION \ + nki==$NKI_VERSION \ + neuron_agentic_development==$NEURON_AGENTIC_DEVELOPMENT_VERSION \ + libtorch-neuronx-lite==$LIBTORCH_NEURONX_LITE_VERSION \ + /opt/vllm-omni-neuron \ + /opt/vllm-neuron \ + && rm -rf ~/.cache/pip/* + +FROM ${BUILD_STAGE} AS final + +# Upgrade OS packages to latest versions +RUN rm -f /etc/apt/sources.list.d/neuron.list /etc/apt/keyrings/neuron.gpg \ + && apt-get update \ + && apt-get upgrade -y \ + && rm -rf /var/lib/apt/lists/* \ + && apt-get clean + +# Point conda at the libmamba solver, then remove the build-time rattler +# solver backend (not needed at runtime). +RUN conda config --set solver libmamba \ + && (conda remove -y --force-remove py-rattler conda-rattler-solver || true) \ + && conda clean -ya \ + && rm -rf /opt/conda/pkgs/* + +EXPOSE 8080 8081 + +ENTRYPOINT ["python", "/usr/local/bin/vllm_entrypoint.py"] +CMD ["/bin/bash"] +HEALTHCHECK CMD curl --fail http://localhost:8080/ping || exit 1