Compare commits
32 Commits
c64127610f
...
eb0d2ecf6d
| Author | SHA1 | Date | |
|---|---|---|---|
| eb0d2ecf6d | |||
| fa6bc9a404 | |||
| 8ad62bcda8 | |||
| 7e8603f91e | |||
| 2d90fc1d90 | |||
| 2e8bef127f | |||
| 29da348bb8 | |||
| 99d8dbe914 | |||
| 672487a72b | |||
| ffbc6d98ff | |||
| 204493bd4c | |||
| 689102e651 | |||
| c8802647b9 | |||
| 341f837e6e | |||
| 2a783325f0 | |||
| 9fcc0eac1c | |||
| a179482018 | |||
| cef614d895 | |||
| 5fe1510981 | |||
| 3d9087c39b | |||
| 301424c1f7 | |||
| 17dbe11ddb | |||
| 84d8d7ce5a | |||
| 1b8bfa7230 | |||
| 649cf0ab45 | |||
| 3f17169d17 | |||
| f79c5099e2 | |||
| 5dfbe9f52e | |||
| aab95c2149 | |||
| 0235d32a8e | |||
| 7b4006b1d4 | |||
| 82c6e04253 |
@@ -0,0 +1,97 @@
|
||||
name: Build & Publish AMD R9700 Toolboxes
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
backends:
|
||||
description: >
|
||||
Comma-separated backends to build (e.g. "rocm-7beta,rocm-7rc").
|
||||
Use "all" to build everything.
|
||||
required: false
|
||||
default: all
|
||||
|
||||
env:
|
||||
DOCKERHUB_REPO: docker.io/kyuz0/amd-r9700-toolboxes
|
||||
LOCAL_PREFIX: llama
|
||||
|
||||
jobs:
|
||||
# 1) Prepare a clean JSON array for the matrix
|
||||
prepare:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
matrix_json: ${{ steps.mk.outputs.matrix_json }}
|
||||
steps:
|
||||
- id: mk
|
||||
shell: bash
|
||||
run: |
|
||||
# Input from the Run workflow form
|
||||
IN='${{ inputs.backends }}'
|
||||
|
||||
if [[ "$IN" == "all" || -z "$IN" ]]; then
|
||||
JSON='["rocm-6.4.4","rocm-7.2","rocm7-nightlies","vulkan-amdvlk","vulkan-radv"]'
|
||||
else
|
||||
# Remove spaces and build JSON array from comma list
|
||||
IN_CLEAN=$(echo "$IN" | tr -d '[:space:]')
|
||||
JSON='["'${IN_CLEAN//,/\",\"}'"]'
|
||||
fi
|
||||
|
||||
echo "matrix_json=${JSON}" >> "$GITHUB_OUTPUT"
|
||||
echo "Using matrix: ${JSON}"
|
||||
|
||||
# 2) Build each backend in parallel using the prepared matrix
|
||||
build-and-push:
|
||||
needs: prepare
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
backend: ${{ fromJson(needs.prepare.outputs.matrix_json) }}
|
||||
|
||||
steps:
|
||||
- name: Free up runner disk space
|
||||
run: |
|
||||
echo "Before cleanup:" && df -h /
|
||||
sudo rm -rf \
|
||||
/usr/share/dotnet \
|
||||
/usr/local/lib/android \
|
||||
/opt/ghc \
|
||||
/opt/hostedtoolcache/CodeQL
|
||||
docker system prune --all --force
|
||||
docker builder prune --all --force
|
||||
echo "After cleanup:" && df -h /
|
||||
|
||||
- name: Check out repository
|
||||
uses: actions/checkout@v3
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
uses: docker/login-action@v2
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Set build timestamp
|
||||
run: echo "BUILD_TS=$(date +%Y%m%dT%H%M%S)" >> $GITHUB_ENV
|
||||
|
||||
- name: Build & push ${{ matrix.backend }}
|
||||
working-directory: toolboxes
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
B="${{ matrix.backend }}"
|
||||
DF="Dockerfile.$B"
|
||||
NAME="${B}"
|
||||
LI="${LOCAL_PREFIX}-${NAME}"
|
||||
TAG="${NAME}_${BUILD_TS}"
|
||||
IMM="${DOCKERHUB_REPO}:${TAG}"
|
||||
CHN="${DOCKERHUB_REPO}:${NAME}"
|
||||
|
||||
echo "→ Building ${DF}"
|
||||
docker build --no-cache -t "${LI}" -f "${DF}" .
|
||||
|
||||
echo "→ Tag & push immutable → ${IMM}"
|
||||
docker tag "${LI}" "${IMM}"
|
||||
docker push "${IMM}"
|
||||
|
||||
echo "→ Tag & push channel → ${CHN}"
|
||||
docker tag "${IMM}" "${CHN}"
|
||||
docker push "${CHN}"
|
||||
@@ -0,0 +1,105 @@
|
||||
name: Poll llama.cpp & Trigger Build
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: '0 */4 * * *'
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
actions: write
|
||||
|
||||
jobs:
|
||||
poll-and-trigger:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- id: fetch
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
REPO_URL="https://github.com/ggml-org/llama.cpp.git"
|
||||
DEFAULT_REF=$(git ls-remote --symref "$REPO_URL" HEAD | awk '/^ref:/ {print $2}')
|
||||
echo "📌 Default branch: ${DEFAULT_REF#refs/heads/}"
|
||||
LATEST_SHA=$(git ls-remote "$REPO_URL" "$DEFAULT_REF" | cut -f1)
|
||||
if [[ -z "$LATEST_SHA" ]]; then echo "❌ No SHA found"; exit 1; fi
|
||||
echo "✅ Latest SHA: $LATEST_SHA"
|
||||
echo "latest_sha=$LATEST_SHA" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- id: previous
|
||||
shell: bash
|
||||
env:
|
||||
GH_REPO: ${{ github.repository }}
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
echo "🔎 Checking for prior artifact 'last-llama-sha'…"
|
||||
ART_ID=$(curl -fsSL -H "Authorization: Bearer $GH_TOKEN" \
|
||||
"https://api.github.com/repos/${GH_REPO}/actions/artifacts?per_page=100" \
|
||||
| jq -r '.artifacts | map(select(.name=="last-llama-sha" and .expired==false)) | sort_by(.created_at) | reverse | .[0].id // empty')
|
||||
if [[ -n "$ART_ID" ]]; then
|
||||
echo "📦 Found artifact id: $ART_ID (downloading)"
|
||||
curl -fsSL -H "Authorization: Bearer $GH_TOKEN" -L \
|
||||
"https://api.github.com/repos/${GH_REPO}/actions/artifacts/${ART_ID}/zip" -o artifact.zip
|
||||
unzip -l artifact.zip || true
|
||||
unzip -p artifact.zip last_commit_sha > last_commit_sha || true
|
||||
else
|
||||
echo "ℹ️ No prior artifact found"
|
||||
fi
|
||||
PREV_SHA=""
|
||||
if [[ -f last_commit_sha ]]; then PREV_SHA=$(cat last_commit_sha); fi
|
||||
echo "🕓 Previous SHA: $PREV_SHA"
|
||||
echo "previous_sha=$PREV_SHA" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- id: compare
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
echo "🧮 Comparing SHAs…"
|
||||
echo "prev: ${{ steps.previous.outputs.previous_sha }}"
|
||||
echo "curr: ${{ steps.fetch.outputs.latest_sha }}"
|
||||
if [[ "${{ steps.fetch.outputs.latest_sha }}" != "${{ steps.previous.outputs.previous_sha }}" ]]; then
|
||||
echo "🔁 New commit detected"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "✅ No change"
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Trigger build_and_publish.yml on main
|
||||
if: steps.compare.outputs.changed == 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
WF="build_and_publish.yml"
|
||||
REF="main"
|
||||
echo "🚀 Dispatching $WF on $REF…"
|
||||
CODE=$(curl -s -o /tmp/resp -w "%{http_code}" \
|
||||
-X POST \
|
||||
-H "Accept: application/vnd.github+json" \
|
||||
-H "Authorization: Bearer $GH_TOKEN" \
|
||||
-d "{\"ref\":\"$REF\",\"inputs\":{\"backends\":\"all\"}}" \
|
||||
"https://api.github.com/repos/${{ github.repository }}/actions/workflows/$WF/dispatches")
|
||||
echo "HTTP $CODE"
|
||||
if [[ "$CODE" != "204" ]]; then echo "Response:"; cat /tmp/resp; exit 1; fi
|
||||
|
||||
|
||||
- name: Save new SHA
|
||||
if: steps.compare.outputs.changed == 'true'
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
printf "%s" "${{ steps.fetch.outputs.latest_sha }}" > last_commit_sha
|
||||
echo "💾 Saved $(wc -c < last_commit_sha) bytes to $(pwd)/last_commit_sha"
|
||||
ls -la last_commit_sha
|
||||
|
||||
- name: Upload last-SHA artifact
|
||||
if: steps.compare.outputs.changed == 'true'
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: last-llama-sha
|
||||
path: last_commit_sha
|
||||
retention-days: 7
|
||||
@@ -0,0 +1,89 @@
|
||||
name: Prune Old Toolbox Images
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
backends:
|
||||
description: Comma-separated backends to prune (e.g. "rocm-7beta,rocm-7rc") or "all"
|
||||
default: all
|
||||
keep:
|
||||
description: Number of latest tags to keep
|
||||
default: "3"
|
||||
workflow_run:
|
||||
workflows: ["Build & Publish AMD Strix Halo Toolboxes"]
|
||||
types: [completed] # runs after success/failure/cancel
|
||||
branches: [main]
|
||||
|
||||
jobs:
|
||||
prune:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
NS: kyuz0
|
||||
REPO: docker.io/kyuz0/amd-r9700-toolboxes
|
||||
steps:
|
||||
- name: Install jq
|
||||
run: sudo apt-get update && sudo apt-get install -y jq
|
||||
|
||||
- name: Login to Docker Hub API (JWT)
|
||||
id: login
|
||||
env:
|
||||
DH_USER: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
DH_PASS: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
run: |
|
||||
TOKEN=$(curl -s -H "Content-Type: application/json" \
|
||||
-d "{\"username\":\"${DH_USER}\",\"password\":\"${DH_PASS}\"}" \
|
||||
https://hub.docker.com/v2/users/login/ | jq -r .token)
|
||||
if [[ -z "$TOKEN" || "$TOKEN" == "null" ]]; then
|
||||
echo "Failed to get Docker Hub JWT"; exit 1
|
||||
fi
|
||||
echo "token=${TOKEN}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Determine backend list
|
||||
id: mk
|
||||
shell: bash
|
||||
run: |
|
||||
IN='${{ github.event.inputs.backends }}'
|
||||
if [[ "$IN" == "all" || -z "$IN" ]]; then
|
||||
JSON='["rocm-6.4.4","rocm-7.2","rocm7-nightlies","vulkan-amdvlk","vulkan-radv"]'
|
||||
else
|
||||
IN_CLEAN=$(echo "$IN" | tr -d '[:space:]')
|
||||
JSON='["'${IN_CLEAN//,/\",\"}'"]'
|
||||
fi
|
||||
echo "list=${JSON}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Prune old tags
|
||||
env:
|
||||
TOKEN: ${{ steps.login.outputs.token }}
|
||||
KEEP: ${{ github.event.inputs.keep }}
|
||||
run: |
|
||||
BACKENDS='${{ steps.mk.outputs.list }}'
|
||||
mapfile -t ARR < <(jq -r '.[]' <<< "$BACKENDS")
|
||||
base_url="https://hub.docker.com/v2/repositories/${NS}/${REPO}/tags"
|
||||
auth_hdr="Authorization: JWT ${TOKEN}"
|
||||
|
||||
for B in "${ARR[@]}"; do
|
||||
echo ""
|
||||
echo "=== Backend: ${B} (keeping latest ${KEEP}) ==="
|
||||
next="${base_url}?page_size=100&ordering=last_updated&name=${B}_"
|
||||
tags=()
|
||||
while [[ -n "$next" && "$next" != "null" ]]; do
|
||||
resp=$(curl -s -H "$auth_hdr" "$next")
|
||||
page_tags=($(jq -r '.results[].name' <<< "$resp" | grep -E "^${B}_" || true))
|
||||
tags+=("${page_tags[@]}")
|
||||
next=$(jq -r '.next' <<< "$resp")
|
||||
done
|
||||
total=${#tags[@]}
|
||||
echo "Found ${total} immutable tag(s) for ${B}."
|
||||
if (( total <= KEEP )); then
|
||||
echo "Nothing to delete."
|
||||
continue
|
||||
fi
|
||||
to_delete=("${tags[@]:KEEP}")
|
||||
for t in "${to_delete[@]}"; do
|
||||
echo "Deleting tag ${t}..."
|
||||
curl -s -X DELETE -H "$auth_hdr" \
|
||||
"https://hub.docker.com/v2/repositories/${NS}/${REPO}/tags/${t}/" \
|
||||
-o /dev/null -w "%{http_code}\n" | grep -Eq "^(202|204)$" \
|
||||
&& echo "✔ Deleted" || echo "✖ Failed"
|
||||
done
|
||||
done
|
||||
@@ -0,0 +1 @@
|
||||
__pycache__
|
||||
Generated
+5
@@ -0,0 +1,5 @@
|
||||
# 默认忽略的文件
|
||||
/shelf/
|
||||
/workspace.xml
|
||||
# 基于编辑器的 HTTP 客户端请求
|
||||
/httpRequests/
|
||||
Generated
+12
@@ -0,0 +1,12 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<module type="PYTHON_MODULE" version="4">
|
||||
<component name="NewModuleRootManager">
|
||||
<content url="file://$MODULE_DIR$" />
|
||||
<orderEntry type="jdk" jdkName="Python 3.13 (base)" jdkType="Python SDK" />
|
||||
<orderEntry type="sourceFolder" forTests="false" />
|
||||
</component>
|
||||
<component name="PyDocumentationSettings">
|
||||
<option name="format" value="PLAIN" />
|
||||
<option name="myDocStringFormat" value="Plain" />
|
||||
</component>
|
||||
</module>
|
||||
+14
@@ -0,0 +1,14 @@
|
||||
<component name="InspectionProjectProfileManager">
|
||||
<profile version="1.0">
|
||||
<option name="myName" value="Project Default" />
|
||||
<inspection_tool class="PyPackageRequirementsInspection" enabled="true" level="WARNING" enabled_by_default="true">
|
||||
<option name="ignoredPackages">
|
||||
<list>
|
||||
<option value="python-jose" />
|
||||
<option value="passlib" />
|
||||
<option value="aiosqlite" />
|
||||
</list>
|
||||
</option>
|
||||
</inspection_tool>
|
||||
</profile>
|
||||
</component>
|
||||
+6
@@ -0,0 +1,6 @@
|
||||
<component name="InspectionProjectProfileManager">
|
||||
<settings>
|
||||
<option name="USE_PROJECT_PROFILE" value="false" />
|
||||
<version value="1.0" />
|
||||
</settings>
|
||||
</component>
|
||||
Generated
+4
@@ -0,0 +1,4 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="ProjectRootManager" version="2" project-jdk-name="Python 3.13 (base)" project-jdk-type="Python SDK" />
|
||||
</project>
|
||||
Generated
+8
@@ -0,0 +1,8 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="ProjectModuleManager">
|
||||
<modules>
|
||||
<module fileurl="file://$PROJECT_DIR$/.idea/amd-r9700-ai-toolboxes.iml" filepath="$PROJECT_DIR$/.idea/amd-r9700-ai-toolboxes.iml" />
|
||||
</modules>
|
||||
</component>
|
||||
</project>
|
||||
Generated
+6
@@ -0,0 +1,6 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="VcsDirectoryMappings">
|
||||
<mapping directory="" vcs="Git" />
|
||||
</component>
|
||||
</project>
|
||||
@@ -1,2 +1,148 @@
|
||||
# amd-r9700-ai-toolboxes
|
||||
# AMD R9700 Llama.cpp Toolboxes
|
||||
|
||||
This project provides pre-built containers (“toolboxes”) for running LLMs on **AMD Radeon AI PRO R9700** GPUs (gfx1201). It uses `toolbox` (standard on Fedora, available on Ubuntu, Arch, etc.) to run `llama.cpp` with full GPU acceleration (Vulkan or ROCm) without messing up your host system.
|
||||
|
||||
## Watch the YouTube Video
|
||||
|
||||
[](https://youtu.be/dgyqBUD71lg)
|
||||
|
||||
## 🚀 Quick Start
|
||||
|
||||
### 1. Create a Toolbox
|
||||
**Which backend to choose?**
|
||||
* **Vulkan (RADV)**: Recommended for **stability**. It works reliably with almost all models.
|
||||
* **ROCm**: Recommended for **maximum performance**.
|
||||
* *Note*: Multiple ROCm versions are available (e.g., 6.4.4, 7.1, 7.9). Performance can vary significantly depending on the model architecture (e.g., Llama vs. Qwen). **Check the [Benchmarks](https://kyuz0.github.io/amd-r9700-ai-toolboxes/)** to find the best version for your model.
|
||||
|
||||
**Option A: Vulkan (RADV) [Recommended]**
|
||||
```bash
|
||||
toolbox create llama-vulkan-radv \
|
||||
--image docker.io/kyuz0/amd-r9700-toolboxes:vulkan-radv \
|
||||
-- --device /dev/dri --group-add video --security-opt seccomp=unconfined
|
||||
```
|
||||
|
||||
**Option B: ROCm (7.2)**
|
||||
```bash
|
||||
toolbox create llama-rocm-7.2 \
|
||||
--image docker.io/kyuz0/amd-r9700-toolboxes:rocm-7.2 \
|
||||
-- --device /dev/dri --device /dev/kfd \
|
||||
--group-add video --group-add render --group-add sudo --security-opt seccomp=unconfined
|
||||
```
|
||||
|
||||
> **Ubuntu Users**: `toolbox` may have issues with GPU access. Use [Distrobox](https://github.com/89luca89/distrobox) instead. See [Detailed Guide](#ubuntu-users-distrobox) below.
|
||||
|
||||
### 2. Enter the Toolbox
|
||||
```bash
|
||||
toolbox enter llama-vulkan-radv
|
||||
# or: toolbox enter llama-rocm-7.2
|
||||
```
|
||||
|
||||
### 3. Download a Model
|
||||
|
||||
**Option A: Manual Download (Recommended)**
|
||||
Use the `hf` tool to download the model GGUF files to a local directory.
|
||||
|
||||
```bash
|
||||
# Download to models/qwen3-coder-30B-A3B/
|
||||
HF_HUB_ENABLE_HF_TRANSFER=1 hf download unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF \
|
||||
BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf \
|
||||
--local-dir .
|
||||
```
|
||||
|
||||
**Multi-shard Models:**
|
||||
If a model is split into multiple files (e.g., `00001-of-00005.gguf`), you must download **all** shards to the same folder. The command above ensures all parts are downloaded.
|
||||
|
||||
> [!NOTE]
|
||||
> The old `huggingface-cli` is deprecated. Use the modern `hf` tool (part of `huggingface_hub`).
|
||||
|
||||
**Option B: Automatic Download (via llama.cpp)**
|
||||
`llama.cpp` can automatically download models from the Hugging Face Hub to its internal cache (`~/.cache/huggingface/hub`).
|
||||
|
||||
```bash
|
||||
# Automatically download and run
|
||||
llama-cli -hf unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF -hf-file BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf ...
|
||||
```
|
||||
*Note: We prefer Option A (dedicated folder) to keep things organized, but Option B is great for quick tests.*
|
||||
|
||||
### 4. Run a Model
|
||||
|
||||
> [!TIP]
|
||||
> You should **always** use `-fa 1` (Flash Attention). This significantly improves performance and memory utilization on the R9700.
|
||||
|
||||
Use **`llama-cli`** for running models directly in your terminal—ideal for quick tests, benchmarking, or chatting without leaving the shell.
|
||||
|
||||
Use **`llama-server`** to start an OpenAI-compatible API server. This allows you to connect third-party UIs (like Open WebUI), use the built-in web interface, or build your own applications using standard libraries.
|
||||
|
||||
**Run it (CLI Chat):**
|
||||
```bash
|
||||
llama-cli -ngl 999 -fa 1 \
|
||||
-m models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf \
|
||||
-p "Write a R9700 toolkit haiku."
|
||||
```
|
||||
|
||||
**Or run as Server (API + Web UI):**
|
||||
```bash
|
||||
llama-server -m models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf \
|
||||
-c 8192 -ngl 999 -fa 1
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 📖 Detailed Guide
|
||||
|
||||
### Managing Toolboxes
|
||||
|
||||
#### Ubuntu Users (Distrobox)
|
||||
If you are on Ubuntu, use Distrobox to ensure proper GPU access:
|
||||
```bash
|
||||
distrobox create -n llama-rocm-7.2 \
|
||||
--image docker.io/kyuz0/amd-r9700-toolboxes:rocm-7.2 \
|
||||
--additional-flags "--device /dev/kfd --device /dev/dri --group-add video --group-add render --security-opt seccomp=unconfined"
|
||||
distrobox enter llama-rocm-7.2
|
||||
```
|
||||
|
||||
#### Updating Toolboxes
|
||||
To pull the latest images and recreate your toolboxes (useful when Llama.cpp updates):
|
||||
```bash
|
||||
# Refresh all toolboxes
|
||||
./refresh-toolboxes.sh all
|
||||
|
||||
# Or refresh specific ones
|
||||
./refresh-toolboxes.sh llama-vulkan-radv llama-rocm-7.2
|
||||
```
|
||||
|
||||
## 📦 Architecture & Containers
|
||||
|
||||
### Backends
|
||||
* **Vulkan**: Cross-platform, very stable.
|
||||
* **RADV (Mesa)**: Best compatibility.
|
||||
* **AMDVLK**: Official AMD driver. Faster in some cases but has a strict 2GB single buffer limit (some large models won't load).
|
||||
* **ROCm**: AMD's compute stack (CUDA-like).
|
||||
|
||||
### Supported Container Images
|
||||
Images are hosted on [Docker Hub](https://hub.docker.com/r/kyuz0/amd-r9700-toolboxes/tags) and automatically rebuilt on Llama.cpp updates.
|
||||
|
||||
| Tag | Backend | Notes |
|
||||
| :--- | :--- | :--- |
|
||||
| `vulkan-radv` | Vulkan (Mesa RADV) | Most stable and compatible. Recommended for most users and all models. |
|
||||
| `vulkan-amdvlk` | Vulkan (AMDVLK) | Fastest backend—AMD open-source driver. ≤2 GiB single buffer allocation limit, some large models won't load. |
|
||||
| `rocm-6.4.4` | ROCm 6.4.4 (Fedora 43) | Latest stable 6.x build. Uses Fedora 43 packages with backported patch for kernel 6.18.4+ support. |
|
||||
| `rocm-7.2` | ROCm 7.2 | Latest stable 7.x build. Includes patch for kernel 6.18.4+ support. |
|
||||
| `rocm7-nightlies` | ROCm 7 Nightlies | Nightly build for ROCm 7. |
|
||||
|
||||
## ⚡ Performance & Planning
|
||||
|
||||
### Benchmarks
|
||||
Check the [Interactive Benchmark Viewer](https://kyuz0.github.io/amd-r9700-ai-toolboxes/) or [docs/benchmarks.md](docs/benchmarks.md) to see performance numbers.
|
||||
|
||||
### VRAM Estimator
|
||||
Use the included script to estimate memory usage for models + context. This helps avoid OOM errors.
|
||||
```bash
|
||||
gguf-vram-estimator.py models/my-model.gguf --contexts 4096 32768
|
||||
```
|
||||
See [docs/vram-estimator.md](docs/vram-estimator.md) for more details.
|
||||
|
||||
## References
|
||||
|
||||
* [Llama.cpp GitHub Repository](https://github.com/ggerganov/llama.cpp)
|
||||
* [AMD RDNA™ 4 Architecture](https://www.amd.com/en/products/graphics/rdna-architecture.html)
|
||||
|
||||
@@ -0,0 +1,300 @@
|
||||
#!/usr/bin/env python3
|
||||
import re, glob, os, json, time
|
||||
from pathlib import Path
|
||||
|
||||
RESULT_SOURCES = [
|
||||
("results", False), # regular single-node runs
|
||||
]
|
||||
OUT_JSON = "../docs/results.json"
|
||||
|
||||
# --- Regexes ---------------------------------------------------------------
|
||||
|
||||
# Table headers come in two shapes (with or without "fa" column)
|
||||
HEADER_RE = re.compile(r"^\|\s*model\s*\|", re.IGNORECASE)
|
||||
SEP_RE = re.compile(r"^\|\s*-+")
|
||||
|
||||
# Build line, e.g. "build: cd6983d5 (6119)"
|
||||
BUILD_RE = re.compile(r"build:\s*([0-9a-f]{7,})\s*\((\d+)\)", re.IGNORECASE)
|
||||
|
||||
# Error classifiers (same spirit as your table script)
|
||||
LOAD_ERR = re.compile(r"failed to load model|Device memory allocation.*failed|⚠️\s*Fail", re.IGNORECASE)
|
||||
HANG_ERR = re.compile(r"GPU Hang|HW Exception", re.IGNORECASE)
|
||||
GENERIC_ERR= re.compile(r"error:|exit \d+|runtime error|⚠️\s*Runtime Error", re.IGNORECASE)
|
||||
|
||||
# Extract numeric ± numeric from the last column
|
||||
TS_RE = re.compile(r"([\d.]+)\s*±\s*([\d.]+)")
|
||||
|
||||
# Quantization from model name
|
||||
QUANT_RE = re.compile(r"(Q\d+_[A-Z0-9_]+|BF16|F16|F32|mxfp\d+)", re.IGNORECASE)
|
||||
|
||||
PARAMS_RE = re.compile(r"([\d.,]+)\s*B", re.IGNORECASE)
|
||||
GIB_RE = re.compile(r"([\d.,]+)\s*GiB", re.IGNORECASE)
|
||||
|
||||
# "30B", "235B" from model name
|
||||
NAME_B_RE = re.compile(r"(\d+(?:\.\d+)?)B")
|
||||
|
||||
# Shard suffix in filenames
|
||||
SHARD_RE = re.compile(r"-000\d+-of-000\d+", re.IGNORECASE)
|
||||
|
||||
# Long-context suffix in filenames (e.g., __longctx32768)
|
||||
LONGCTX_RE = re.compile(r"longctx(\d+)", re.IGNORECASE)
|
||||
|
||||
# --- Helpers ---------------------------------------------------------------
|
||||
|
||||
ENV_CANON = {
|
||||
"rocm7_1": "rocm7.1",
|
||||
"rocm7_alpha": "rocm-7alpha",
|
||||
}
|
||||
|
||||
def clean_model_name(raw):
|
||||
base = SHARD_RE.sub("", raw)
|
||||
return base
|
||||
|
||||
def canonicalize_env(env):
|
||||
if not env:
|
||||
return env
|
||||
for raw, canon in ENV_CANON.items():
|
||||
prefix = f"{raw}-"
|
||||
if env == raw:
|
||||
return canon
|
||||
if env.startswith(prefix):
|
||||
return canon + env[len(raw):]
|
||||
return env
|
||||
|
||||
def parse_env_flags(basename):
|
||||
"""
|
||||
pattern: <model>__<env>[__fa1][__hblt0][__longctx32768][__rpc][__single][__dual]
|
||||
Returns (env, fa, context_tag, context_tokens, rpc_flag, gpu_config)
|
||||
"""
|
||||
parts = basename.split("__")
|
||||
if len(parts) < 2:
|
||||
return None, False, "default", None, False, "single"
|
||||
|
||||
env = parts[1]
|
||||
fa = False
|
||||
context_tag = "default"
|
||||
context_tokens = None
|
||||
rpc_flag = False
|
||||
gpu_config = "single" # default to single if not specified
|
||||
|
||||
for raw_suffix in parts[2:]:
|
||||
suffix = raw_suffix.lower()
|
||||
if suffix == "fa1":
|
||||
fa = True
|
||||
elif suffix == "hblt0":
|
||||
env = f"{env}-hblt0"
|
||||
elif suffix.startswith("longctx"):
|
||||
context_tag = suffix
|
||||
m = LONGCTX_RE.search(suffix)
|
||||
if m:
|
||||
try:
|
||||
context_tokens = int(m.group(1))
|
||||
except ValueError:
|
||||
context_tokens = None
|
||||
elif suffix == "rpc":
|
||||
rpc_flag = True
|
||||
elif suffix == "single":
|
||||
gpu_config = "single"
|
||||
elif suffix == "dual":
|
||||
gpu_config = "dual"
|
||||
|
||||
return env, fa, context_tag, context_tokens, rpc_flag, gpu_config
|
||||
|
||||
def env_base_and_variant(env):
|
||||
# e.g. "rocm6_4_2-rocwmma" -> ("rocm6_4_2", "rocwmma")
|
||||
if "-" in env:
|
||||
base, variant = env.split("-", 1)
|
||||
return base, variant
|
||||
return env, None
|
||||
|
||||
def detect_error(text):
|
||||
if LOAD_ERR.search(text):
|
||||
return True, "load"
|
||||
if HANG_ERR.search(text):
|
||||
return True, "hang"
|
||||
if GENERIC_ERR.search(text):
|
||||
return True, "runtime"
|
||||
return False, None
|
||||
|
||||
def parse_table(text):
|
||||
"""
|
||||
Returns list of rows parsed from the markdown-like table.
|
||||
Each row is a dict of the parsed columns, normalized by header names.
|
||||
Handles presence/absence of the 'fa' column.
|
||||
"""
|
||||
lines = text.splitlines()
|
||||
rows = []
|
||||
header = None
|
||||
col_idx = {}
|
||||
|
||||
for i, line in enumerate(lines):
|
||||
if HEADER_RE.search(line):
|
||||
# header line
|
||||
header = [c.strip().lower() for c in line.strip().strip("|").split("|")]
|
||||
# next line should be the separator; skip it
|
||||
# build index map
|
||||
for idx, name in enumerate(header):
|
||||
col_idx[name] = idx
|
||||
continue
|
||||
if header and (SEP_RE.search(line) or not line.strip()):
|
||||
# skip separators / blanks after header
|
||||
continue
|
||||
if header and line.startswith("|"):
|
||||
parts = [c.strip() for c in line.strip().strip("|").split("|")]
|
||||
# guard for short lines
|
||||
if len(parts) < len(header):
|
||||
continue
|
||||
row = {}
|
||||
for name, idx in col_idx.items():
|
||||
row[name] = parts[idx]
|
||||
rows.append(row)
|
||||
# stop parsing block when a blank line after some rows appears
|
||||
if header and line.strip() == "" and rows:
|
||||
break
|
||||
|
||||
return rows
|
||||
|
||||
def coerce_float(m, default=None):
|
||||
try:
|
||||
return float(m)
|
||||
except:
|
||||
return default
|
||||
|
||||
def extract_quant(model_name):
|
||||
m = QUANT_RE.search(model_name)
|
||||
return (m.group(1).upper() if m else None)
|
||||
|
||||
def b_from_name(model_name):
|
||||
m = NAME_B_RE.search(model_name)
|
||||
return coerce_float(m.group(1)) if m else None
|
||||
|
||||
# --- Main scan -------------------------------------------------------------
|
||||
|
||||
runs = []
|
||||
builds = set()
|
||||
envs = set()
|
||||
|
||||
for results_dir, is_rpc_source in RESULT_SOURCES:
|
||||
glob_pattern = os.path.join(results_dir, "*.log")
|
||||
for path in sorted(glob.glob(glob_pattern)):
|
||||
base = os.path.basename(path).rsplit(".log", 1)[0]
|
||||
if "__" not in base:
|
||||
continue
|
||||
|
||||
model_raw, _rest = base.split("__", 1)
|
||||
env, fa_from_name, context_tag, context_tokens, rpc_flag, gpu_config = parse_env_flags(base)
|
||||
env = canonicalize_env(env)
|
||||
if env:
|
||||
envs.add(env)
|
||||
|
||||
model_clean = clean_model_name(model_raw)
|
||||
|
||||
with open(path, errors="ignore") as f:
|
||||
text = f.read()
|
||||
|
||||
# build info (take the last match in file if many)
|
||||
build_hash, build_num = None, None
|
||||
for m in BUILD_RE.finditer(text):
|
||||
build_hash, build_num = m.group(1), m.group(2)
|
||||
if build_hash:
|
||||
builds.add((build_hash, build_num))
|
||||
|
||||
# detect error (if there is no valid table rows)
|
||||
table_rows = parse_table(text)
|
||||
|
||||
# If table rows exist, we’ll still mark errors only if no perf found
|
||||
has_pp = any(r.get("test","").lower()=="pp512" for r in table_rows)
|
||||
has_tg = any(r.get("test","").lower()=="tg128" for r in table_rows)
|
||||
error, etype = (False, None)
|
||||
if not (has_pp or has_tg):
|
||||
error, etype = detect_error(text)
|
||||
|
||||
# Determine FA flag:
|
||||
# prefer explicit column "fa" if present, else fallback to filename "__fa1"
|
||||
fa_in_table = None
|
||||
for r in table_rows:
|
||||
if "fa" in r:
|
||||
try:
|
||||
fa_in_table = int(r["fa"]) == 1
|
||||
except:
|
||||
fa_in_table = None
|
||||
break
|
||||
fa_enabled = fa_in_table if fa_in_table is not None else fa_from_name
|
||||
|
||||
# Normalize env base / variant (e.g., rocwmma)
|
||||
env_base, env_variant = env_base_and_variant(env)
|
||||
|
||||
# Emit one run per row (pp512 / tg128)
|
||||
for r in table_rows or [{}]:
|
||||
test = r.get("test", "").lower() if table_rows else None
|
||||
tps_mean, tps_std = None, None
|
||||
if table_rows:
|
||||
ts_field = r.get("t/s", "")
|
||||
m = TS_RE.search(ts_field)
|
||||
if m:
|
||||
tps_mean = coerce_float(m.group(1))
|
||||
tps_std = coerce_float(m.group(2))
|
||||
|
||||
# parse numeric helpers from row (if present)
|
||||
params_b = None
|
||||
file_size_gib = None
|
||||
if "params" in r:
|
||||
pm = PARAMS_RE.search(r["params"])
|
||||
if pm:
|
||||
params_b = coerce_float(pm.group(1).replace(",", ""))
|
||||
if "size" in r:
|
||||
sm = GIB_RE.search(r["size"])
|
||||
if sm:
|
||||
file_size_gib = coerce_float(sm.group(1).replace(",", ""))
|
||||
|
||||
# quant from model name (unchanged)
|
||||
quant = extract_quant(model_clean)
|
||||
|
||||
# name_params_b: prefer table value; else fall back to B in model name
|
||||
name_params_b = params_b if params_b is not None else b_from_name(model_clean)
|
||||
|
||||
backend = r.get("backend")
|
||||
ngl = r.get("ngl")
|
||||
mmap = r.get("mmap")
|
||||
|
||||
run = {
|
||||
"model": model_raw,
|
||||
"model_clean": model_clean,
|
||||
"env": env,
|
||||
"env_base": env_base,
|
||||
"env_variant": env_variant, # e.g. "rocwmma"
|
||||
"fa": bool(fa_enabled),
|
||||
"context": context_tag or "default",
|
||||
"context_tokens": context_tokens,
|
||||
"test": test, # "pp512" | "tg128" | None (if error)
|
||||
"tps_mean": tps_mean,
|
||||
"tps_std": tps_std,
|
||||
"error": bool(error),
|
||||
"error_type": etype, # "load" | "hang" | "runtime" | None
|
||||
"backend": backend,
|
||||
"ngl": (int(ngl) if (ngl and ngl.isdigit()) else None),
|
||||
"mmap": (int(mmap) if (mmap and mmap.isdigit()) else None),
|
||||
"params_b": params_b, # from table, if available
|
||||
"file_size_gib": file_size_gib, # from table, if available
|
||||
"name_params_b": name_params_b, # parsed from model name (e.g., 30B -> 30.0)
|
||||
"quant": quant,
|
||||
"log": path,
|
||||
"rpc": bool(is_rpc_source or rpc_flag),
|
||||
"gpu_config": gpu_config,
|
||||
"build": {"hash": build_hash, "number": build_num} if build_hash else None,
|
||||
}
|
||||
runs.append(run)
|
||||
|
||||
# Meta
|
||||
meta = {
|
||||
"generated_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
||||
"os_kernel": "Fedora 42 — Linux 6.15.9-201.fc42.x86_64 (Sat Aug 2 11:37:34 UTC 2025)",
|
||||
"llamacpp_builds": [{"hash": h, "number": n} for (h, n) in sorted(builds)],
|
||||
"environments": sorted(envs),
|
||||
"notes": "pp512 = prompt processing; tg128 = text generation; t/s = tokens/second",
|
||||
}
|
||||
|
||||
out = {"meta": meta, "runs": runs}
|
||||
|
||||
Path(OUT_JSON).write_text(json.dumps(out, indent=2))
|
||||
print(f"Wrote {OUT_JSON} with {len(runs)} rows.")
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 251.12 ± 0.62 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 7.53 ± 0.03 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 124.60 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.52 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm-7-nightly-rocwmma] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 246.03 ± 0.16 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 7.41 ± 0.02 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 123.04 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.46 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm-7-nightly-rocwmma] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 239.85 ± 0.14 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 7.37 ± 0.01 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 99.59 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.82 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm-7-nightly] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 240.74 ± 0.07 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 7.41 ± 0.01 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 100.10 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.83 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm-7-nightly] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 235.80 ± 0.18 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 7.01 ± 0.02 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 95.03 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.09 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm-7.9-rocwmma] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 233.71 ± 0.16 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 6.94 ± 0.01 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 94.33 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.09 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm-7.9-rocwmma] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 235.23 ± 0.14 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 6.94 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 98.01 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.45 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm-7.9] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 237.08 ± 0.15 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 7.00 ± 0.01 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 99.40 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.47 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm-7.9] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 235.42 ± 0.14 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 7.02 ± 0.02 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 94.65 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.09 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm6_4_4-rocwmma] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 233.43 ± 0.03 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 6.94 ± 0.01 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 94.89 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.09 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm6_4_4-rocwmma] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 238.15 ± 0.10 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 6.99 ± 0.01 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 109.79 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.50 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm6_4_4] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 239.15 ± 0.10 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 7.03 ± 0.02 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 109.76 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.48 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm6_4_4] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 236.00 ± 0.17 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 7.01 ± 0.02 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 94.52 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.09 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm7.1.1-rocwmma] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 233.56 ± 0.08 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 6.93 ± 0.01 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 94.48 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.09 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm7.1.1-rocwmma] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 235.60 ± 0.08 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 6.93 ± 0.01 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 99.54 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.47 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm7.1.1] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__fa1 __longctx32768 failed (exit 0)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | pp512 | 237.37 ± 0.19 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 1 | tg128 | 7.01 ± 0.02 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 99.62 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 6.45 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 2 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
main: error: failed to create context with model '/home/kyuz0/models/Q3_K_S/Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002.gguf'
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
✖ ! [rocm7.1.1] Devstral-2-123B-Instruct-2512-Q3_K_S-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0)
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
ggml_vulkan: Found 3 Vulkan devices:
|
||||
ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | pp512 | 133.82 ± 0.17 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | tg128 | 7.55 ± 0.01 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
ggml_vulkan: Found 3 Vulkan devices:
|
||||
ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 44.89 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | tg32 @ d16384 | 4.37 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
ggml_vulkan: Found 3 Vulkan devices:
|
||||
ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 16.45 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | tg32 @ d32768 | 3.13 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
ggml_vulkan: Found 3 Vulkan devices:
|
||||
ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none
|
||||
ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | pp512 | 104.90 ± 0.13 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | tg128 | 8.35 ± 0.01 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
ggml_vulkan: Found 3 Vulkan devices:
|
||||
ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none
|
||||
ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 74.48 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | tg32 @ d16384 | 6.44 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
WARNING: radv is not a conformant Vulkan implementation, testing use only.
|
||||
ggml_vulkan: Found 3 Vulkan devices:
|
||||
ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none
|
||||
ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 36.00 ± 0.00 |
|
||||
| llama ?B Q3_K - Small | 50.63 GiB | 125.03 B | Vulkan | 99 | 1 | tg32 @ d32768 | 2.94 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 964.08 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 19.68 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 571.14 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 17.45 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | pp512 | 2843.52 ± 16.68 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | tg128 | 22.18 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 305.90 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 19.69 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 252.57 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 17.46 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | pp512 | 380.51 ± 0.29 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | tg128 | 22.17 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 757.63 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 20.22 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 449.05 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 18.64 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | pp512 | 2767.12 ± 15.29 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | tg128 | 22.21 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 279.80 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 20.23 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 222.90 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 18.62 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | pp512 | 380.81 ± 0.21 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | tg128 | 22.21 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 731.41 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 20.15 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 419.35 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 18.45 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | pp512 | 2608.48 ± 2.57 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | tg128 | 22.28 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 271.63 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 20.15 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 213.61 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 18.44 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | pp512 | 368.45 ± 0.14 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | tg128 | 22.29 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 761.85 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 20.30 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 447.32 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 18.69 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | pp512 | 2644.29 ± 2.60 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | tg128 | 22.32 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 274.70 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 20.30 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 219.43 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 18.70 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | pp512 | 370.52 ± 0.20 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | tg128 | 22.32 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 719.86 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 20.22 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 421.46 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 18.48 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | pp512 | 2405.48 ± 3.53 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | tg128 | 22.36 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 276.54 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 20.21 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 217.88 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 18.48 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | pp512 | 378.57 ± 0.27 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | tg128 | 22.35 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 828.88 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 20.37 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | n_ubatch | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 501.73 ± 0.00 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 18.74 ± 0.00 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
@@ -0,0 +1,10 @@
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
|
||||
ggml_cuda_init: found 1 ROCm devices:
|
||||
Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32
|
||||
| model | size | params | backend | ngl | fa | test | t/s |
|
||||
| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | pp512 | 2468.22 ± 5.84 |
|
||||
| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 1 | tg128 | 22.39 ± 0.01 |
|
||||
|
||||
build: 2aa45ef9e (7423)
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user