# requirements.txt
#
# Pinned to exact versions verified working together on an NVIDIA H100 80GB
# (driver 570.211.01, CUDA 12.8) so `pip install -r requirements.txt` gives a
# reproducible install on any similarly-equipped CUDA machine.
#
# torch: provides the CUDA/bf16 tensor runtime that every agent's model.generate()
# call depends on. Version 2.2.2 was confirmed to report
# torch.cuda.is_available() == True and torch.cuda.is_bf16_supported() == True
# on the target GPU, which is the numeric precision the token-injection
# generate() path relies on.
torch==2.2.2

# transformers: supplies AutoTokenizer/AutoModelForCausalLM and, critically, the
# `input_ids=` kwarg on `.generate()` -- this is the exact mechanism that lets
# the downstream agents bypass their own CPU tokenizer. Pinned to the version
# already verified (4.57.6) because a newer/older release could change
# chat-template or generate() default kwargs.
transformers==4.57.6

# numpy: backs utils/token_manager.py's `.npy` save/load of the token-ID arrays
# that cross the /dev/shm boundary between agents. `.npy` is used (rather than
# raw struct packing) because it self-describes dtype/shape in its header,
# eliminating a whole class of byte-order/misalignment failure that a
# hand-rolled binary format could otherwise introduce silently.
numpy==1.26.4

# pyyaml: parses/emits the YAML frontmatter block that carries the OKF
# `token_pointer` metadata key (and sibling fields) between agents in
# utils/okf_parser.py. Chosen over hand-rolled string parsing because YAML's
# block-scalar and list syntax must round-trip exactly for `tags: [...]` to
# survive multiple agent read/write cycles without drifting.
pyyaml==6.0.3

# accelerate: transformers' `AutoModelForCausalLM.from_pretrained(...,
# low_cpu_mem_usage=True)` call in utils/model_loader.py delegates its
# streamed-weight-loading implementation to this package -- without it,
# `low_cpu_mem_usage=True` either errors or silently falls back to
# materializing a full fp32 copy of the weights in host RAM before casting,
# defeating the exact peak-RAM optimization that kwarg is there for.
accelerate==1.14.0
