Add LLM compose stack for phy-srv-gpu01 (not deployed yet)
Follows the homelab pattern: ironicbadger.docker_compose_generator v2 renders services/<host>/NN-<stack>/compose.yml templates into ~/docker/compose.yaml on the host. - 01-vllm: chat model, fixed --gpu-memory-utilization - 02-embeddings: second vLLM instance (--task embed) rather than a separate toolchain, so SM120 support only has to be solved once - 03-openwebui: Open WebUI + pgvector (not chroma — corpus size) - 99-network: shared bridge; leading comment keeps networks: top-level - pin docker_compose_generator to 2.0.1 — galaxy tags mix v1/v2 formats - group_vars: stack config incl. LDAP placeholders still to be filled The role only writes the compose file; starting the stack stays manual. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,37 @@
|
||||
services:
|
||||
# Embeddings run on a second vLLM instance rather than a separate toolchain
|
||||
# (e.g. text-embeddings-inference): whatever vLLM build works on SM120 then
|
||||
# covers embeddings too, instead of having to solve Blackwell support twice.
|
||||
embeddings:
|
||||
image: "{{ vllm_image }}"
|
||||
container_name: embeddings
|
||||
networks:
|
||||
- llmnet
|
||||
ports:
|
||||
- "8001:8000"
|
||||
volumes:
|
||||
- "{{ appdata_path }}/models/huggingface:/root/.cache/huggingface"
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=all
|
||||
- NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||||
- "HUGGING_FACE_HUB_TOKEN={{ hf_token | default('') }}"
|
||||
command:
|
||||
- --model
|
||||
- "{{ embedding_model }}"
|
||||
- --served-model-name
|
||||
- "{{ embedding_model_name }}"
|
||||
- --task
|
||||
- embed
|
||||
# small fixed slice — the chat model gets the rest
|
||||
- --gpu-memory-utilization
|
||||
- "{{ embeddings_gpu_memory_utilization }}"
|
||||
ipc: host
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: 1
|
||||
capabilities: [gpu]
|
||||
runtime: nvidia
|
||||
restart: unless-stopped
|
||||
Reference in New Issue
Block a user