feat: add MTP (Multi-Token Prediction) support via new ROCm 7.2.3 and Vulkan RADV toolboxes

2026-05-14 20:09:03 +01:00
parent 7320eb3f00
commit 3e3f3674a8
4 changed files with 186 additions and 1 deletions
@@ -0,0 +1,68 @@
+# build stage
+FROM registry.fedoraproject.org/fedora:43 AS builder
+
+# deps
+RUN dnf -y --nodocs --setopt=install_weak_deps=False install \
+  git vim \
+  make gcc cmake ninja-build lld clang clang-devel compiler-rt libcurl-devel \
+  vulkan-loader-devel vulkaninfo mesa-vulkan-drivers \
+  spirv-headers-devel radeontop glslc patch \
+  && dnf clean all && rm -rf /var/cache/dnf/*
+
+# llama.cpp (am17an mtp-clean fork — Multi-Token Prediction)
+WORKDIR /opt/llama.cpp
+RUN git clone -b mtp-clean --single-branch https://github.com/am17an/llama.cpp.git .
+
+COPY llama-grammar.patch /tmp/llama-grammar.patch
+
+# build
+RUN git clean -xdf \
+  && git submodule update --recursive \
+  && patch -p1 < /tmp/llama-grammar.patch \
+  && cmake -S . -B build -G Ninja \
+  -DGGML_VULKAN=ON \
+  -DCMAKE_BUILD_TYPE=Release \
+  -DGGML_RPC=ON \
+  -DCMAKE_INSTALL_PREFIX=/usr \
+  -DLLAMA_BUILD_TESTS=OFF \
+  -DLLAMA_BUILD_EXAMPLES=ON \
+  -DLLAMA_BUILD_SERVER=ON \
+  && cmake --build build --config Release \
+  && cmake --install build --config Release
+
+# libs
+RUN find /opt/llama.cpp/build -type f -name 'lib*.so*' -exec cp {} /usr/lib64/ \; \
+  && ldconfig
+
+# helper
+COPY gguf-vram-estimator.py /usr/local/bin/gguf-vram-estimator.py
+RUN chmod +x /usr/local/bin/gguf-vram-estimator.py
+
+
+# runtime stage
+FROM registry.fedoraproject.org/fedora-minimal:43
+
+# runtime deps
+RUN microdnf -y --nodocs --setopt=install_weak_deps=0 install \
+  bash ca-certificates libatomic libstdc++ libgcc \
+  vulkan-loader vulkan-loader-devel vulkaninfo mesa-vulkan-drivers radeontop procps-ng \
+  && microdnf clean all && rm -rf /var/cache/dnf/*
+
+# copy
+COPY --from=builder /usr/ /usr/
+COPY --from=builder /usr/local/ /usr/local/
+COPY --from=builder /opt/llama.cpp/build/bin/rpc-* /usr/local/bin/
+
+# ld
+RUN echo "/usr/local/lib"  > /etc/ld.so.conf.d/local.conf \
+  && echo "/usr/local/lib64" >> /etc/ld.so.conf.d/local.conf \
+  && ldconfig \
+  && cp -n /usr/local/lib/libllama*.so* /usr/lib64/ 2>/dev/null || true \
+  && ldconfig
+
+# helper
+COPY gguf-vram-estimator.py /usr/local/bin/gguf-vram-estimator.py
+RUN chmod +x /usr/local/bin/gguf-vram-estimator.py
+
+# shell
+CMD ["/bin/bash"]