Compare commits

..

6 Commits

Author SHA1 Message Date
dab1ecfdd7 Merge remote-tracking branch 'origin/master' into local_changes
Some checks failed
Build Hermes agent / build (pull_request) Has been cancelled
Build ollama (gfx906) / build (pull_request) Has been cancelled
# Conflicts:
#	ai/compose.yml
2026-07-07 15:04:00 -04:00
e595381173 Merge pull request 'feat: add llama-cpp-hermes service with ROCm 6.1 + gfx906 support' (#54) from feat/hermes43-llamacpp into master
Some checks failed
Build Hermes agent / build (push) Has been cancelled
Build ollama (gfx906) / build (push) Has been cancelled
Reviewed-on: #54
2026-07-07 19:00:03 +00:00
54e0661396 fix: add missing USER hermes at end of Dockerfile
The Dockerfile was switching to USER root for the final chown but never
switched back to USER hermes. This caused ALL container processes to run
as root (uid 0) instead of the hermes user (uid 10000).

The entrypoint's gosu privilege drop only caught the main exec chain,
leaving backgrounded subprocesses (dashboard PTY sessions, workers with
start_new_session=True) running as root — creating files owned by root
in ExoKortex and breaking Syncthing sync.

Adding USER hermes at the end ensures the container runs unprivileged
and ALL child processes inherit uid 10000 from the start.
2026-07-07 14:58:44 -04:00
1e290410de Merge branch 'master' into feat/hermes43-llamacpp
Some checks failed
Build Hermes agent / build (pull_request) Has been cancelled
Build ollama (gfx906) / build (pull_request) Has been cancelled
Conflicts resolved:
- hermes env: keep OLLAMA_HOST=ollama-cpu + master's new vars
- ollama→ollama-cpu rename: keep our rename over master's ollama
2026-07-07 14:50:59 -04:00
80c9906757 fix: rename ollama→ollama-cpu, fix llama-cpp-hermes YAML indentation
Some checks failed
Build Hermes agent / build (pull_request) Has been cancelled
Build ollama (gfx906) / build (pull_request) Has been cancelled
- Rename ollama service to ollama-cpu (CPU-only for bge-m3 embeddings)
- Fix llama-cpp-hermes indented under networks instead of as top-level service
- Update hermes OLLAMA_HOST to point to ollama-cpu

Part of PR #54 GPU/ROCm refactor
2026-07-07 14:29:23 -04:00
3c92d93366 feat: add llama-cpp-hermes service with ROCm 6.1 + gfx906 support
Some checks failed
Build Hermes agent / build (pull_request) Has been cancelled
Build ollama (gfx906) / build (pull_request) Has been cancelled
- Add custom llama.cpp Dockerfile with ROCm 6.1 + gfx906 (MI50) build
- Add llama-cpp-hermes service serving Hermes 4.3 on dual MI50 GPUs
- Strip GPU devices/ROCm env from ollama service (CPU-only for embeddings)

Hermes 4.3 runs at ~19 t/s on dual MI50s with 160K context.
2026-06-11 11:41:42 -04:00
3 changed files with 67 additions and 54 deletions

View File

@@ -15,7 +15,7 @@ services:
environment: environment:
- HERMES_UID=10000 - HERMES_UID=10000
- HERMES_GID=10000 - HERMES_GID=10000
- OLLAMA_HOST=http://ollama:11434 - OLLAMA_HOST=http://ollama-cpu:11434
- HERMES_DASHBOARD=1 - HERMES_DASHBOARD=1
# Multi-profile: comma-separated list of profiles to run as gateways. # Multi-profile: comma-separated list of profiles to run as gateways.
# The entrypoint reads this and starts one gateway per profile. # The entrypoint reads this and starts one gateway per profile.
@@ -108,12 +108,12 @@ services:
- "traefik.http.services.syncthing.loadbalancer.server.port=8384" - "traefik.http.services.syncthing.loadbalancer.server.port=8384"
ollama: ollama-cpu:
build: build:
context: ./ollama context: ./ollama
dockerfile: Dockerfile dockerfile: Dockerfile
image: ollama/ollama:rocm-gfx906 image: ollama/ollama:rocm-gfx906
container_name: ollama container_name: ollama-cpu
tty: true tty: true
restart: always restart: always
ports: ports:
@@ -124,22 +124,42 @@ services:
- /mnt/HoardingCow_docker_data/Ollama/ollama:/root/.ollama - /mnt/HoardingCow_docker_data/Ollama/ollama:/root/.ollama
environment: environment:
- OLLAMA_VULKAN=0 - OLLAMA_VULKAN=0
- HSA_OVERRIDE_GFX_VERSION=9.0.6
- HCC_AMDGPU_TARGET=gfx906
- HIP_VISIBLE_DEVICES=0,1
- ROCR_VISIBLE_DEVICES=0,1
- HSA_ENABLE_SDMA=0
- OLLAMA_HOST=0.0.0.0 - OLLAMA_HOST=0.0.0.0
- OLLAMA_DEBUG=1
- OLLAMA_FLASH_ATTENTION=1 llama-cpp-hermes:
- OLLAMA_NUM_PARALLEL=2 image: llama-cpp:rocm-gfx906
container_name: llama-cpp-hermes
restart: unless-stopped
networks:
- ai_backend
ports:
- "127.0.0.1:8300:8080"
ipc: host
devices: devices:
# Map the render nodes and KFD for ROCm to work inside the container
- /dev/kfd:/dev/kfd - /dev/kfd:/dev/kfd
- /dev/dri:/dev/dri - /dev/dri:/dev/dri
group_add: group_add:
- "303" - "303"
- "26" - "26"
environment:
- HSA_OVERRIDE_GFX_VERSION=9.0.6
- HSA_ENABLE_SDMA=0
- HIP_VISIBLE_DEVICES=0,1
- LLAMA_CACHE=/models
volumes:
- /mnt/HoardingCow_docker_data/Llama_cpp/models:/models
- /mnt/HoardingCow_docker_data/Ollama/ollama/models/blobs/sha256-17823599694fa3503ef54bf748d5078c6ce881f4d01616cafa255dc05d215a08:/model.gguf:ro
command: >
-m /model.gguf
--host 0.0.0.0
--port 8080
--gpu-layers 99
--ctx-size 163840
-ctk f16 -ctv f16
--flash-attn on
--split-mode layer
--no-mmap
--n-predict -1
# --- Honcho + OpenConcho combiné: API + Web UI nginx/FastAPI --- # --- Honcho + OpenConcho combiné: API + Web UI nginx/FastAPI ---
honcho: honcho:
@@ -228,48 +248,6 @@ volumes:
driver: bridge driver: bridge
name: honcho_data name: honcho_data
# llama_cpp_devstral:
# image: ghcr.io/ggml-org/llama.cpp:server-rocm
# container_name: llama_cpp_devstral
# restart: unless-stopped
# networks:
# - ai_backend
# ports:
# - "8300:8080"
# ipc: host
# devices:
# - "/dev/kfd:/dev/kfd"
# - "/dev/dri:/dev/dri"
# group_add:
# - "303" # video
# - "26" # render
# environment:
# HSA_OVERRIDE_GFX_VERSION: 9.0.6
# HIP_VISIBLE_DEVICES: 0,1
# LLAMA_CACHE: /models
# volumes:
# - /mnt/HoardingCow_docker_data/Llama_cpp/models:/models
# - /mnt/HoardingCow_docker_data/Llama_cpp/devstral-agent.jinja:/template.jinja
# command: >
# -hf unsloth/Devstral-Small-2-24B-Instruct-2512-GGUF:Devstral-Small-2-24B-Instruct-2512-Q8_0.gguf
# -a devstral-2-small-llama_cpp
# --chat-template-file /template.jinja
# --host 0.0.0.0
# --port 8080
# --n-gpu-layers 99
# --ctx-size 163840
# --batch-size 4096
# --ubatch-size 4096
# --cache-type-k f16
# --cache-type-v f16
# --cache-reuse 256
# --flash-attn on
# --context-shift
# --split-mode layer
# --no-mmap
# --n-predict -1
# --parallel 2
# vllm: # vllm:
# image: nalanzeyu/vllm-gfx906:v0.9.0-rocm6.3 # image: nalanzeyu/vllm-gfx906:v0.9.0-rocm6.3
# container_name: vllm # container_name: vllm

View File

@@ -75,3 +75,8 @@ USER root
RUN chown -R hermes:hermes /opt/hermes/tools /opt/hermes/toolsets.py RUN chown -R hermes:hermes /opt/hermes/tools /opt/hermes/toolsets.py
VOLUME [ "/opt/data" ] VOLUME [ "/opt/data" ]
# Switch to the hermes user so the container runs unprivileged.
# All child processes inherit this user — no more orphan root processes
# that escape gosu privilege drops in the entrypoint.
USER hermes

30
ai/llama-cpp/Dockerfile Normal file
View File

@@ -0,0 +1,30 @@
# llama-cpp-rocm6/Dockerfile
# Custom llama.cpp server with ROCm 6.1 + gfx906 (MI50) support.
# Build: docker build -t llama-cpp:rocm-gfx906 .
FROM rocm/dev-ubuntu-22.04:6.1.2-complete AS builder
RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y curl git build-essential pkg-config cmake make && rm -rf /var/lib/apt/lists/*
ARG LLAMACPP_VERSION=b9596
RUN git clone --depth 1 --branch ${LLAMACPP_VERSION} https://github.com/ggml-org/llama.cpp.git /build
WORKDIR /build
ENV HIP_PATH=/opt/rocm ROCM_PATH=/opt/rocm PATH=/opt/rocm/bin:/opt/rocm/llvm/bin:${PATH} CMAKE_PREFIX_PATH=/opt/rocm
RUN mkdir build && cd build && \
cmake .. -DGGML_HIP=ON -DCMAKE_BUILD_TYPE=Release \
-DAMDGPU_TARGETS="gfx906:xnack-" \
-DCMAKE_POSITION_INDEPENDENT_CODE=ON \
-DGGML_CUDA=OFF -DGGML_VULKAN=OFF -DGGML_METAL=OFF \
-DBUILD_SHARED_LIBS=OFF && \
cmake --build . --target llama-server -- -j $(nproc)
FROM ubuntu:24.04
RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y \
ca-certificates curl libstdc++6 libgomp1 libopenblas0 \
libnuma1 libelf1 libdrm2 libdrm-amdgpu1 \
&& rm -rf /var/lib/apt/lists/*
COPY --from=builder /opt/rocm/lib/ /opt/rocm/lib/
COPY --from=builder /opt/rocm/share/ /opt/rocm/share/
COPY --from=builder /build/build/bin/llama-server /usr/local/bin/llama-server
RUN echo /opt/rocm/lib > /etc/ld.so.conf.d/rocm.conf && ldconfig
ENV HSA_OVERRIDE_GFX_VERSION=9.0.6 HCC_AMDGPU_TARGET=gfx906 HSA_ENABLE_SDMA=0
EXPOSE 8080
ENTRYPOINT ["/usr/local/bin/llama-server"]