# =============================================================================
# THREE TARGETS. Which one a container gets is chosen by `target:` in
# docker-compose.yml — never by accident.
#
#   base     internal only: OS + pinned Python deps. Not runnable (no source).
#   ml       internal only: base + CPU torch/transformers + the vendored HHEM
#            checkpoint. Not runnable (no source).
#   dev      ml + source + git.  Tests, bench, the bake-off harness.
#   runtime  base + source.      `api` and `worker`. No ML, no git.
#
# B7, the reason this file is staged at all. Every service used to build the
# ONE image (`build: .` three times over), so vendoring a 442 MB checkpoint for
# a BENCH-ONLY scorer took the image the API and the Celery worker actually run
# from 1.28 GB to 2.74 GB. Neither of them imports torch, transformers or
# anything under `bench/`.
#
# READ OFF `docker images` AFTER THE BUILD, not estimated (the first draft of
# this table WAS estimated and every cell in it was wrong):
#
#     before   one shared image       2.74 GB   api, worker AND dev
#     after    api                     985 MB   -1.76 GB
#              worker                  985 MB   -1.76 GB
#              dev                     2.52 GB  -0.22 GB
#
# Reconciliation, so the deltas are checkable rather than asserted:
#   runtime loses torch+transformers (~1.02 GB) and the HHEM checkpoint
#   (442 MB) by never entering the `ml` stage, and both targets lose a
#   duplicated 259 MB layer (see the leaf-tail comment). `dev` shrinks in spite
#   of GAINING git (+37 MB) and the tiktoken table (+2 MB) because that 259 MB
#   duplicate went away. Note `runtime` (985 MB) is now smaller than the
#   1.28 GB single-stage image that predates HHEM entirely.
#
# STAGE ORDER IS LOAD-BEARING: `runtime` is LAST on purpose, because a bare
# `docker build .` with no `--target` builds the last stage (verified, not
# assumed: it produces the 985 MB image). The lean image is what you get by
# default; the 2.5 GB one has to be asked for by name. If you add a stage, add
# it ABOVE `runtime`.
#
# BUILD-CACHE NOTE for whoever edits this next: everything above the torch
# install is a shared prefix, and inserting ANY instruction into it (even an
# `EXPOSE`) invalidates ~1.5 GB of torch layers and forces the 442 MB
# checkpoint to be downloaded again — measured at ~2.5 min. That is why
# `EXPOSE` and the user-creation live in the leaf stages and are duplicated
# there rather than hoisted into `base`. Pay that cost deliberately, not by
# accident.
# =============================================================================

FROM python:3.12-slim AS base

ENV PYTHONDONTWRITEBYTECODE=1 \
    PYTHONUNBUFFERED=1 \
    PIP_DISABLE_PIP_VERSION_CHECK=1 \
    PIP_NO_CACHE_DIR=1

WORKDIR /app

RUN apt-get update && apt-get install -y --no-install-recommends \
      build-essential \
      curl \
    && rm -rf /var/lib/apt/lists/*

# Install dependencies first for layer caching. We need `src/` to exist for
# the editable build backend to find packages, but we keep the actual code
# COPY in a separate (later) layer so dep changes don't bust source caches.
COPY pyproject.toml ./
COPY src/customgpt_indexer/__init__.py ./src/customgpt_indexer/__init__.py
RUN pip install --upgrade pip && pip install -e ".[dev]"

# ---------------------------------------------------------------------------
# Vendor the tiktoken BPE table, for the same reason the HHEM checkpoint is
# vendored further down — found by running the dev image under
# `docker run --network none`, which is the only way this class of bug shows
# itself:
#
#     requests.exceptions.ConnectionError: HTTPSConnectionPool(
#       host='openaipublic.blob.core.windows.net', port=443) ...
#       /encodings/cl100k_base.tiktoken
#
# `agent/tools.py` line 17 is `_ENC = tiktoken.get_encoding("cl100k_base")` at
# MODULE SCOPE, so this download happens on IMPORT, in every container, with an
# empty cache every time (tiktoken's default cache is a temp dir, which does
# not survive `--rm`). Consequences, in order of how much they cost:
#   * `api` and `worker` cannot BOOT if that blob is unreachable — an outage at
#     Microsoft becomes an outage here, for a 1.7 MB constant.
#   * a bake-off arm inherits a network dependency on the path that computes
#     token counts, i.e. truncation budgets and cost. An arm that dies on a DNS
#     blip reads as a worse arm.
# It fails loudly rather than silently (the import raises; nothing falls back
# to an approximate token count), which is why this was a latent fragility and
# not a wrong number. Baking it makes the tokenizer as reproducible as the
# weights: identical bytes, no network, every run.
#
# The cache is root-owned and the container runs as `appuser`, so a request for
# an encoding that is NOT baked here (only `cl100k_base` is used today) fails
# with a permission error instead of quietly reaching the internet. If you add
# one, add it to this line.
# ---------------------------------------------------------------------------
ENV TIKTOKEN_CACHE_DIR=/opt/tiktoken-cache
RUN python -c "import tiktoken; tiktoken.get_encoding('cl100k_base')" \
 && python -c "import os; assert os.listdir('/opt/tiktoken-cache'), 'tiktoken cache is empty'"

# =============================================================================
# `ml` — the bake-off's ML stack. Reached ONLY by the `dev` target.
# =============================================================================
FROM base AS ml

# ---------------------------------------------------------------------------
# Bake-off: Vectara HHEM-2.1-Open, the deterministic hallucination scorer.
#
# (1) CPU-ONLY TORCH. `pip install torch` from PyPI on linux/amd64 resolves to
#     the CUDA build and drags in ~2.5 GB of nvidia-* wheels (cudnn, cublas,
#     nccl, ...) that this image can never use — there is no GPU in CI, in the
#     dev container, or in the bake-off harness. The `+cpu` wheel index is the
#     supported way to say so; it is ~1 GB installed instead of ~3.5 GB.
#     `--index-url` (not `--extra-index-url`) so resolution cannot silently
#     fall back to the CUDA build on a mirror hiccup; the CPU index carries
#     torch's own dependencies too.
#
# (2) SEPARATE LAYER, AFTER the app deps, so the ~1.5 GB of torch + weights is
#     cached independently of `pyproject.toml` churn.
#
#     MEASURED COST when it was added, on the single-stage image this file
#     replaced: 1.28 GB -> 2.74 GB (+1.46 GB), of which ~1.02 GB is
#     torch+transformers and 442 MB is the checkpoint.
#
#     That cost is now confined to ONE target. The caveat this comment used to
#     carry — "api, worker and dev all build this one image, so the API and the
#     Celery worker carry an ML runtime they never import" — was the B7
#     blocker, and it is fixed by the `runtime` stage at the bottom of this
#     file, which forks off `base` and never sees this layer.
# ---------------------------------------------------------------------------
RUN pip install --index-url https://download.pytorch.org/whl/cpu "torch>=2.4,<3.0" \
 && pip install "transformers>=4.44,<5.0" "huggingface-hub>=0.25,<1.0"

# ---------------------------------------------------------------------------
# (3) VENDOR THE WEIGHTS INTO THE IMAGE, at a pinned commit SHA.
#
# The alternative — lazy `from_pretrained()` on first score — makes a
# DETERMINISTIC scorer depend on huggingface.co being up in the middle of a
# benchmark run, and tracks branch HEAD so a Vectara push silently changes
# every number. See `_vendor_hhem.py` for the full argument.
#
# `--smoke` runs an OFFLINE forward pass inside the build. That is what turns
# "the files are present" into "no code path needs the network", including the
# `AutoConfig.from_pretrained("google/flan-t5-base")` hidden inside Vectara's
# trust_remote_code module. If it fails, the BUILD fails — never the run.
# ---------------------------------------------------------------------------
ENV HF_HOME=/opt/hf-cache
COPY src/customgpt_indexer/bench/scorers/_vendor_hhem.py /tmp/_vendor_hhem.py
#
# The `chmod` is in THIS layer, not a later one, on purpose: a `RUN chmod -R`
# in its own layer rewrites the metadata of every file it touches and so
# duplicates the entire 442 MB cache as a second layer. Measured: +442 MB of
# pure waste. Folding it here costs nothing.
#
# Read-only and root-owned: the scorer runs as `appuser`, which must be able
# to READ the checkpoint and must not be able to REPLACE it. A writable cache
# would let anything running in the container swap the weights while the
# pinned SHA in `hhem.py` kept reading as honoured — the pin has to mean
# something. It also stops `transformers` from trying to take cache lock files
# at score time, which fails confusingly under an unwritable HOME.
RUN python /tmp/_vendor_hhem.py --smoke \
 && rm /tmp/_vendor_hhem.py \
 && chown -R root:root /opt/hf-cache \
 && chmod -R a-w,a+rX /opt/hf-cache

# Scoring time performs ZERO network I/O, enforced by the environment rather
# than by convention. `hhem.py` records the observed value as
# `details["offline"]` in every scorecard, and `tests/test_hhem_scorer.py`
# asserts it under LIVE_HHEM=1 — so if this line is ever dropped, the loss of
# network isolation shows up as a failing test and a changed provenance field,
# not as a mystery stall three hours into a run. Override only to re-vendor.
ENV HF_HUB_OFFLINE=1 \
    TRANSFORMERS_OFFLINE=1 \
    HF_HUB_DISABLE_TELEMETRY=1

# =============================================================================
# `dev` — tests, bench, bake-off. The only target that can produce a scorecard.
# =============================================================================
FROM ml AS dev

# ---------------------------------------------------------------------------
# B6a: `git`. Without it EVERY scorecard this repo can emit is stamped
# non-citable for a reason that has nothing to do with the numbers.
#
# `bench/runner.py::_git_provenance()` shells out to `git rev-parse HEAD` and
# `git status --porcelain`. There was no git binary here, so it fell through to
# hand-parsing `.git/HEAD` — which was also not mounted — and ended at
# `git_sha_unknown`, an unconditional blocker, which makes `provenance.citable`
# permanently False on every run.
#
# WHY THE BINARY AND NOT A BUILD-ARG SHA. A `--build-arg GIT_SHA=$(git rev-parse
# HEAD)` baked into an ENV is a snapshot of the moment the image was built, and
# this compose file bind-mounts `./src` READ-WRITE into the container. The code
# that runs is the host worktree as it is RIGHT NOW, not as it was at build
# time. A baked sha would therefore go stale silently and would report a
# committed-looking sha for code that has since changed — the "confident wrong
# number" failure mode, applied to provenance itself. Worse, a baked sha cannot
# express dirtiness at all: it renders uncommitted work as a clean commit id.
# So: a real binary, reading the real worktree, at run time.
#
# The binary alone is not enough — it also needs a repo to look at. That is the
# `- .:/repo:ro` mount plus GIT_DIR/GIT_WORK_TREE on the `dev` service in
# docker-compose.yml; see the long comment there for why the worktree examined
# must be the HOST tree and not `/app`.
#
# COST: +37.3 MB, measured off `docker history` on THIS image — not the 99 MB
# git costs on a bare python:3.12-slim, because `base` already installs
# build-essential, which drags in the perl that most of git's weight is.
# Paid ONLY by this stage. `runtime` does not get git — see the note at the
# bottom of this file.
# ---------------------------------------------------------------------------
RUN apt-get update && apt-get install -y --no-install-recommends \
      git \
    && rm -rf /var/lib/apt/lists/*

# The ownership check git performs during repository DISCOVERY ("detected
# dubious ownership") fires when the repo's owner uid differs from the caller's
# — the normal case in CI, where the GitHub runner checks out as uid 1001 and
# this container runs as uid 1000. `_git_provenance()` itself is immune because
# GIT_DIR is set explicitly (verified: uid 1001 against a uid-1000 worktree
# still returns the correct sha), but a human typing `cd /repo && git log` is
# not, and a provenance tool nobody can reproduce by hand is a bad tool.
# Scoped to the two paths a repo can appear at, not `*`.
RUN git config --system --add safe.directory /repo \
 && git config --system --add safe.directory /app

# ---- leaf tail. KEEP IN SYNC with the identical block in `runtime`. ---------
# (Duplicated rather than hoisted into `base`: hoisting it would insert layers
#  into the cached prefix and cost a 1.5 GB rebuild. See the header.)
# S7 fix: non-root user. RCE in any dependency now lands as `appuser` rather
# than root. UID 1000 picked to match typical host user so bind-mount writes
# don't break. Created BEFORE the COPY so `COPY --chown` can name it, which is
# not cosmetic: the trailing `RUN chown -R appuser:appuser /app` this replaces
# rewrote the metadata of every file it touched, and a metadata rewrite
# copies-up the whole tree into a second layer. Measured on `docker history`:
# `COPY . .` 259 MB followed by `chown -R` 259 MB — half a gigabyte per target
# to set an owner bit. Same trap as the HHEM `chmod`, documented above.
RUN groupadd --system --gid 1000 appuser \
 && useradd --system --uid 1000 --gid appuser --shell /bin/bash appuser \
 && chown appuser:appuser /app

COPY --chown=appuser:appuser . .

# Re-install editable so the freshly-copied source is picked up.
RUN pip install -e .

EXPOSE 8000

# =============================================================================
# `runtime` — `api` and `worker`. LAST STAGE ON PURPOSE (see header).
#
# Forks off `base`, so it never inherits torch, transformers or the 442 MB
# checkpoint: 2.74 GB -> 985 MB.
#
# NO GIT HERE, DELIBERATELY. Neither service produces a scorecard (`_git_
# provenance()` has exactly one caller, `bench/runner.py`, and the bench CLI
# runs in `dev`), and `/repo` is not mounted into them either, so the 37 MB
# would buy nothing. If someone ever DOES run the bench from one of these
# containers, the provenance collector degrades the way it is designed to —
# `git_sha_unknown: no `git` binary on PATH`, a blocker, `citable: false`. It
# reports the absence; it does not invent a sha. That is the correct failure
# and it is why this omission is safe to make.
# =============================================================================
FROM base AS runtime

# ---- leaf tail. KEEP IN SYNC with the identical block in `dev`. -------------
# S7 fix: non-root user. RCE in any dependency now lands as `appuser` rather
# than root. UID 1000 picked to match typical host user so bind-mount writes
# don't break. Created BEFORE the COPY so `COPY --chown` can name it, which is
# not cosmetic: the trailing `RUN chown -R appuser:appuser /app` this replaces
# rewrote the metadata of every file it touched, and a metadata rewrite
# copies-up the whole tree into a second layer. Measured on `docker history`:
# `COPY . .` 259 MB followed by `chown -R` 259 MB — half a gigabyte per target
# to set an owner bit. Same trap as the HHEM `chmod`, documented above.
RUN groupadd --system --gid 1000 appuser \
 && useradd --system --uid 1000 --gid appuser --shell /bin/bash appuser \
 && chown appuser:appuser /app

COPY --chown=appuser:appuser . .

# Re-install editable so the freshly-copied source is picked up.
RUN pip install -e .

EXPOSE 8000
