diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..7dbd58c --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 959.74 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 17.59 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..38b3995 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] Ministral-3-14B-Instruct-2512-BF16__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..8128c4f --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 305.03 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 17.59 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..2118d45 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] Ministral-3-14B-Instruct-2512-BF16__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx16384__single.log new file mode 100644 index 0000000..9cd35d2 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 761.86 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 18.02 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx32768__single.log new file mode 100644 index 0000000..b0371cd --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] Ministral-3-14B-Instruct-2512-BF16__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..bee66c7 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 280.13 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 18.01 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..2fa6209 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] Ministral-3-14B-Instruct-2512-BF16__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..a404cf8 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 665.78 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 17.95 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..a27b19a --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] Ministral-3-14B-Instruct-2512-BF16__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..0746a3b --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 262.83 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 17.95 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..0df8f46 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] Ministral-3-14B-Instruct-2512-BF16__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__fa1__longctx16384__single.log new file mode 100644 index 0000000..cec6fd9 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 767.43 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 18.07 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__fa1__longctx32768__single.log new file mode 100644 index 0000000..f0c41c9 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] Ministral-3-14B-Instruct-2512-BF16__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..30dc5f8 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 275.13 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 18.07 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..4954077 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] Ministral-3-14B-Instruct-2512-BF16__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..8a4d9c0 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 688.43 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 17.97 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..d75180d --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4-rocwmma] Ministral-3-14B-Instruct-2512-BF16__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..8354a36 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 271.52 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 17.98 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..4a10c94 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4-rocwmma] Ministral-3-14B-Instruct-2512-BF16__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx16384__single.log new file mode 100644 index 0000000..4d94d3b --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 829.03 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 18.10 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx32768__single.log new file mode 100644 index 0000000..d2d7db0 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4] Ministral-3-14B-Instruct-2512-BF16__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..620498b --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 290.54 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 18.11 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..4a8744e --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4] Ministral-3-14B-Instruct-2512-BF16__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..95296bb --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 665.80 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 17.95 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..6fcc282 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1-rocwmma] Ministral-3-14B-Instruct-2512-BF16__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..a21f222 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 262.76 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 17.95 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..3aa3f49 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1-rocwmma] Ministral-3-14B-Instruct-2512-BF16__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx16384__single.log new file mode 100644 index 0000000..8d7c867 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 766.84 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 18.09 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx32768__single.log new file mode 100644 index 0000000..794a240 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1] Ministral-3-14B-Instruct-2512-BF16__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..af16c36 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 274.75 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 18.07 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..56550cb --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,8 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/Ministral-3-14B-Instruct-2512-BF16.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1] Ministral-3-14B-Instruct-2512-BF16__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx16384__single.log new file mode 100644 index 0000000..f535030 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 310.96 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | Vulkan | 99 | 1 | tg32 @ d16384 | 13.87 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx32768__single.log new file mode 100644 index 0000000..1cd4977 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 194.73 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | Vulkan | 99 | 1 | tg32 @ d32768 | 10.53 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_radv__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_radv__fa1__longctx16384__single.log new file mode 100644 index 0000000..340e047 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_radv__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 550.05 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | Vulkan | 99 | 1 | tg32 @ d16384 | 15.14 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_radv__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_radv__fa1__longctx32768__single.log new file mode 100644 index 0000000..6075187 --- /dev/null +++ b/benchmark/results/Ministral-3-14B-Instruct-2512-BF16__vulkan_radv__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 351.12 ± 0.00 | +| mistral3 14B BF16 | 25.16 GiB | 13.51 B | Vulkan | 99 | 1 | tg32 @ d32768 | 13.32 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..2ad7367 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1210.41 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 26.73 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..994690a --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 708.36 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 22.94 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..af8ee2e --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 461.89 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 26.74 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..74951c4 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 363.06 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 22.91 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx16384__single.log new file mode 100644 index 0000000..00b41ad --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 958.03 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.59 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx32768__single.log new file mode 100644 index 0000000..294db19 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 552.93 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.80 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..72548b9 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 413.39 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.58 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..a44bbdb --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 313.00 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.79 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..fbfc732 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 832.67 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.47 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..5bf1229 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 459.52 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.45 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..69e1d71 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 384.05 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.46 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..a826ce1 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 282.23 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.46 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__fa1__longctx16384__single.log new file mode 100644 index 0000000..d880d9b --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 963.98 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.76 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__fa1__longctx32768__single.log new file mode 100644 index 0000000..8febd35 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 552.83 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.91 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..c69dd75 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 406.44 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.71 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..fa28713 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 308.45 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.90 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..0ba7dcf --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 868.42 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.51 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..d6d92a8 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 494.21 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.48 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..2c6281a --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 397.69 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.47 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..d4a17cb --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 295.42 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.47 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx16384__single.log new file mode 100644 index 0000000..9281c1e --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1058.69 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.78 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx32768__single.log new file mode 100644 index 0000000..85cb010 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 621.76 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.94 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..8b63ece --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 433.90 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.78 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..80d0fbf --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 336.61 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.93 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..b17543c --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 832.53 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.47 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..2ce4c0b --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 463.80 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.48 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..96d3ab2 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 383.90 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.46 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..929c0b0 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 281.71 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.45 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx16384__single.log new file mode 100644 index 0000000..d0217b4 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 964.51 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.76 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx32768__single.log new file mode 100644 index 0000000..8fee9c9 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 544.22 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.91 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..263fe02 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 406.04 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 27.77 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..9bb937f --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 308.59 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.92 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx16384__single.log new file mode 100644 index 0000000..eedfbf2 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 403.34 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | Vulkan | 99 | 1 | tg32 @ d16384 | 18.97 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx32768__single.log new file mode 100644 index 0000000..60aebef --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 242.85 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | Vulkan | 99 | 1 | tg32 @ d32768 | 14.10 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_radv__fa1__longctx16384__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_radv__fa1__longctx16384__single.log new file mode 100644 index 0000000..9775da1 --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_radv__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 718.51 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | Vulkan | 99 | 1 | tg32 @ d16384 | 22.17 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_radv__fa1__longctx32768__single.log b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_radv__fa1__longctx32768__single.log new file mode 100644 index 0000000..47982eb --- /dev/null +++ b/benchmark/results/Ministral-3-8B-Instruct-2512-BF16__vulkan_radv__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 442.10 ± 0.00 | +| mistral3 8B BF16 | 15.81 GiB | 8.49 B | Vulkan | 99 | 1 | tg32 @ d32768 | 18.81 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..0e9220f --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..c87b02f --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..5f84bac --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..2844ad6 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx16384__dual.log new file mode 100644 index 0000000..71fcd2f --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx32768__dual.log new file mode 100644 index 0000000..9798e23 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..3d7d5bd --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..46e2f23 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..8f39404 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..f620982 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..1c78fed --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..5eea508 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__fa1__longctx16384__dual.log new file mode 100644 index 0000000..5cde01c --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__fa1__longctx32768__dual.log new file mode 100644 index 0000000..851c151 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..82083a4 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..0e62560 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..1e63b91 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..043d690 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..5ffe43c --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..4c63ecd --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__fa1__longctx16384__dual.log new file mode 100644 index 0000000..00b2f1a --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__fa1__longctx32768__dual.log new file mode 100644 index 0000000..abedad7 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..fa3baa8 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..b56b3e4 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..e82205c --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..adc5d6f --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..ea15478 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..31d4cb1 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1-rocwmma] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__fa1__longctx16384__dual.log new file mode 100644 index 0000000..569d23d --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__fa1__longctx32768__dual.log new file mode 100644 index 0000000..a8b38fa --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..29df7c3 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..74c6c43 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/qwen3-coder-30B-A3B/BF16/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1] Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx16384__dual.log new file mode 100644 index 0000000..ac9ab14 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx16384__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B BF16 | 56.89 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 158.35 ± 0.00 | +| qwen3moe 30B.A3B BF16 | 56.89 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d16384 | 12.09 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx32768__dual.log new file mode 100644 index 0000000..e02b1f5 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx32768__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B BF16 | 56.89 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 108.88 ± 0.00 | +| qwen3moe 30B.A3B BF16 | 56.89 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d32768 | 12.02 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_radv__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_radv__fa1__longctx16384__dual.log new file mode 100644 index 0000000..4e6abc9 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_radv__fa1__longctx16384__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B BF16 | 56.89 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 594.90 ± 0.00 | +| qwen3moe 30B.A3B BF16 | 56.89 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d16384 | 13.86 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_radv__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_radv__fa1__longctx32768__dual.log new file mode 100644 index 0000000..a60f383 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_radv__fa1__longctx32768__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B BF16 | 56.89 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 285.47 ± 0.00 | +| qwen3moe 30B.A3B BF16 | 56.89 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d32768 | 9.77 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..08dc232 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1009.52 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 69.94 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..12a8943 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 556.12 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 50.17 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..2bb38fe --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1009.56 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 69.86 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..95a8727 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 559.45 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 51.14 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__fa1__longctx16384__single.log new file mode 100644 index 0000000..d8be377 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 745.53 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 78.82 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__fa1__longctx32768__single.log new file mode 100644 index 0000000..8eed428 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 408.05 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 63.42 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__hblt0__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..cd9c3bd --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 743.47 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 78.71 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__hblt0__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..937fbe4 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 409.20 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 63.48 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..9d556bd --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 624.67 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 72.28 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..fabe2e2 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 336.79 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 56.71 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..26f9d80 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 625.56 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 72.41 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..4391874 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 341.42 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 53.72 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__fa1__longctx16384__single.log new file mode 100644 index 0000000..811856f --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 730.98 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 73.70 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__fa1__longctx32768__single.log new file mode 100644 index 0000000..a408d3f --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 397.89 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 59.19 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__hblt0__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..a164a59 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 736.16 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 73.30 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__hblt0__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..b8e6097 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 406.18 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 59.22 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..e165180 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 668.97 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 73.51 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..b5a7435 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 364.36 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 57.28 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..ba2a5a1 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 669.72 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 73.74 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..61d7b97 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 366.16 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 56.94 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__fa1__longctx16384__single.log new file mode 100644 index 0000000..4213327 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 837.05 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 75.93 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__fa1__longctx32768__single.log new file mode 100644 index 0000000..98ff437 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 471.22 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 60.49 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__hblt0__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..9ade6bf --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 844.53 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 76.50 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__hblt0__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..c57aa7a --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 477.45 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 61.07 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..4cf1e4c --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 622.46 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 72.32 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..ae88ee9 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 341.42 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 56.64 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..779d552 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 623.52 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 71.99 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..4428cba --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 341.60 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 56.65 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__fa1__longctx16384__single.log new file mode 100644 index 0000000..1a051e0 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 733.54 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 73.55 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__fa1__longctx32768__single.log new file mode 100644 index 0000000..ef78548 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 395.85 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 59.13 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__hblt0__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..0d93f78 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 736.46 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 73.45 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__hblt0__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..33fcfe6 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 406.36 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 59.16 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_amdvlk__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_amdvlk__fa1__longctx16384__single.log new file mode 100644 index 0000000..a6e75bc --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_amdvlk__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 389.88 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d16384 | 49.62 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_amdvlk__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_amdvlk__fa1__longctx32768__single.log new file mode 100644 index 0000000..8b8a0ae --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_amdvlk__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 201.27 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d32768 | 23.92 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_radv__fa1__longctx16384__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_radv__fa1__longctx16384__single.log new file mode 100644 index 0000000..bef14d2 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_radv__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 591.92 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d16384 | 64.30 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_radv__fa1__longctx32768__single.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_radv__fa1__longctx32768__single.log new file mode 100644 index 0000000..8e72c84 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_radv__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 332.93 ± 0.00 | +| qwen3moe 30B.A3B Q4_K - Medium | 17.35 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d32768 | 46.15 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..91a816c --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 925.95 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 56.08 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..1ca34a9 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 496.65 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 43.58 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..347ead2 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 880.38 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 56.20 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..1c4f86f --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 518.19 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 43.57 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx16384__dual.log new file mode 100644 index 0000000..0f82d8d --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 700.17 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 61.73 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx32768__dual.log new file mode 100644 index 0000000..316e41c --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 323.72 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 52.03 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..4e46c06 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 674.08 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 61.71 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..023f7f4 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 387.83 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 52.17 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..bdc1263 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 629.39 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 59.02 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..d416d9a --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 342.35 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 49.03 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..41c1079 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 608.61 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 59.46 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..1eaf3d7 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 335.81 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 49.03 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__fa1__longctx16384__dual.log new file mode 100644 index 0000000..72cf6e0 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 737.67 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 59.93 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__fa1__longctx32768__dual.log new file mode 100644 index 0000000..88fdf3d --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 407.90 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 50.80 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..a97dcd4 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 711.88 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 59.70 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..d907db4 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 399.61 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 50.76 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..e43c4e3 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 669.98 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 59.94 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..9571dc7 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 368.96 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 49.40 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..26b3b4d --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 645.77 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 59.74 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..31c7c8e --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 360.83 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 49.26 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__fa1__longctx16384__dual.log new file mode 100644 index 0000000..c649452 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 846.10 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 61.46 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__fa1__longctx32768__dual.log new file mode 100644 index 0000000..5eec4e5 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 477.88 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 52.14 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..5b8372e --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 814.93 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 61.38 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..a4dd497 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 466.97 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 51.78 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..a61cb46 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 625.60 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 59.35 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..c6f5102 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 342.65 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 48.97 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..f035139 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 608.95 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 59.43 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..798e3c8 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 336.21 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 49.08 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__fa1__longctx16384__dual.log new file mode 100644 index 0000000..3567d74 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 742.91 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 59.63 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__fa1__longctx32768__dual.log new file mode 100644 index 0000000..5641586 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 408.78 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 50.72 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..beda5c9 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 713.08 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 59.80 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..7e34ba3 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,11 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 400.11 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 48.06 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx16384__dual.log new file mode 100644 index 0000000..b518aa9 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx16384__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 403.05 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d16384 | 41.38 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx32768__dual.log new file mode 100644 index 0000000..d220933 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx32768__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 207.11 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d32768 | 25.15 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_radv__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_radv__fa1__longctx16384__dual.log new file mode 100644 index 0000000..6e8fbf1 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_radv__fa1__longctx16384__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 602.70 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d16384 | 53.51 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_radv__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_radv__fa1__longctx32768__dual.log new file mode 100644 index 0000000..dba40d5 --- /dev/null +++ b/benchmark/results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_radv__fa1__longctx32768__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 347.37 ± 0.00 | +| qwen3moe 30B.A3B Q8_0 | 33.51 GiB | 30.53 B | Vulkan | 99 | 1 | tg32 @ d32768 | 41.56 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..db76b96 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fb00d261565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fb00d26192b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fb00d261aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7fb00d262921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7fb00d2680ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7fb0107d27e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7fb0107d5468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7fb0107d590e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7fb01070d0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7fb01069f9fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7fb0106a27ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7fb0106a3250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7fb00cbf75b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fb00cbf7668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..5706e7a --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f13c3cd6565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f13c3cd692b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f13c3cd6aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f13c3cd7921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f13c3cdd0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f13c72477e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f13c724a468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f13c724a90e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f13c71820cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f13c71149fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f13c71177ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f13c7118250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f13c366c5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f13c366c668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..c1108ca --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fa627a02565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fa627a0292b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fa627a02aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7fa627a03921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7fa627a090ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7fa62af737e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7fa62af76468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7fa62af7690e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7fa62aeae0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7fa62ae409fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7fa62ae437ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7fa62ae44250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7fa6273985b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fa627398668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..9881881 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f4cf8afa565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f4cf8afa92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f4cf8afaaaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f4cf8afb921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f4cf8b010ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f4cfc06b7e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f4cfc06e468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f4cfc06e90e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f4cfbfa60cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f4cfbf389fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f4cfbf3b7ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f4cfbf3c250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f4cf84905b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f4cf8490668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__fa1__longctx16384__dual.log new file mode 100644 index 0000000..fc1b00a --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fa54139e565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fa54139e92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fa54139eaaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7fa54139f921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7fa5413a50ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7fa5449fc7e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7fa5449ff468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7fa5449ff90e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7fa5449370cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7fa5448c99fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7fa5448cc7ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7fa5448cd250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7fa540d345b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fa540d34668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__fa1__longctx32768__dual.log new file mode 100644 index 0000000..9dd10ef --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f39855b5565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f39855b592b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f39855b5aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f39855b6921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f39855bc0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f3988c137e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f3988c16468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f3988c1690e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f3988b4e0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f3988ae09fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f3988ae37ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f3988ae4250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f3984f4b5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f3984f4b668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..7b0f387 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fe0b4e63565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fe0b4e6392b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fe0b4e63aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7fe0b4e64921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7fe0b4e6a0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7fe0b84c17e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7fe0b84c4468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7fe0b84c490e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7fe0b83fc0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7fe0b838e9fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7fe0b83917ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7fe0b8392250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7fe0b47f95b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fe0b47f9668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..7f61214 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f1cbbefc565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f1cbbefc92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f1cbbefcaaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f1cbbefd921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f1cbbf030ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f1cbf55a7e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f1cbf55d468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f1cbf55d90e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f1cbf4950cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f1cbf4279fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f1cbf42a7ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f1cbf42b250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f1cbb8925b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f1cbb892668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..cb82b1c --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f87ff42a565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f87ff42a92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f87ff42aaaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f87ff42b921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f87ff4310ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f8802be47e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f8802be7468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f8802be790e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f8802b1f0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f8802ab19fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f8802ab47ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f8802ab5250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f87fedc05b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f87fedc0668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7.9-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..6937c82 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f0367d9b565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f0367d9b92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f0367d9baaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f0367d9c921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f0367da20ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f036b5557e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f036b558468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f036b55890e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f036b4900cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f036b4229fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f036b4257ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f036b426250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f03677315b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f0367731668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7.9-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..df3a9e3 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f2073135565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f207313592b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f2073135aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f2073136921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f207313c0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f20768ef7e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f20768f2468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f20768f290e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f207682a0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f20767bc9fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f20767bf7ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f20767c0250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f2072acb5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f2072acb668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7.9-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..3e65d5a --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fde55443565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fde5544392b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fde55443aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7fde55444921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7fde5544a0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7fde58bfd7e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7fde58c00468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7fde58c0090e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7fde58b380cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7fde58aca9fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7fde58acd7ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7fde58ace250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7fde54dd95b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fde54dd9668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7.9-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__fa1__longctx16384__dual.log new file mode 100644 index 0000000..374e318 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f47da474565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f47da47492b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f47da474aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f47da475921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f47da47b0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f47ddd057e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f47ddd08468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f47ddd0890e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f47ddc400cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f47ddbd29fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f47ddbd57ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f47ddbd6250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f47d9e0a5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f47d9e0a668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7.9] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__fa1__longctx32768__dual.log new file mode 100644 index 0000000..113a3ce --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f28b3a40565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f28b3a4092b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f28b3a40aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f28b3a41921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f28b3a470ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f28b72d17e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f28b72d4468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f28b72d490e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f28b720c0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f28b719e9fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f28b71a17ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f28b71a2250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f28b33d65b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f28b33d6668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7.9] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..da38526 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fb1e0233565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fb1e023392b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fb1e0233aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7fb1e0234921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7fb1e023a0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7fb1e3ac47e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7fb1e3ac7468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7fb1e3ac790e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7fb1e39ff0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7fb1e39919fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7fb1e39947ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7fb1e3995250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7fb1dfbc95b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fb1dfbc9668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7.9] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..8f5d6be --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f3b12105565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f3b1210592b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f3b12105aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f3b12106921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f3b1210c0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f3b159967e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f3b15999468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f3b1599990e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f3b158d10cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f3b158639fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f3b158667ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f3b15867250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f3b11a9b5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f3b11a9b668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7.9] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..5ebc023 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f718060f565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f718060f92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f718060faaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f7180610921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f71806160ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f71839b47e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f71839b7468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f71839b790e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f71838ef0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f71838819fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f71838847ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f7183885250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f717ffa55b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f717ffa5668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..45a8554 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f0863e61565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f0863e6192b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f0863e61aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f0863e62921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f0863e680ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f08672067e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f0867209468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f086720990e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f08671410cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f08670d39fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f08670d67ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f08670d7250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f08637f75b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f08637f7668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..b9e8d78 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f3daba19565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f3daba1992b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f3daba19aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f3daba1a921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f3daba200ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f3daedbe7e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f3daedc1468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f3daedc190e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f3daecf90cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f3daec8b9fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f3daec8e7ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f3daec8f250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f3dab3af5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f3dab3af668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..ab06fe2 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7ffbfa658565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7ffbfa65892b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7ffbfa658aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7ffbfa659921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7ffbfa65f0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7ffbfd9fd7e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7ffbfda00468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7ffbfda0090e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7ffbfd9380cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7ffbfd8ca9fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7ffbfd8cd7ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7ffbfd8ce250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7ffbf9fee5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7ffbf9fee668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__fa1__longctx16384__dual.log new file mode 100644 index 0000000..75c18b8 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7faeb5f37565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7faeb5f3792b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7faeb5f37aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7faeb5f38921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7faeb5f3e0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7faeb93617e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7faeb9364468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7faeb936490e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7faeb929c0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7faeb922e9fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7faeb92317ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7faeb9232250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7faeb58cd5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7faeb58cd668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__fa1__longctx32768__dual.log new file mode 100644 index 0000000..a114e6e --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fe280c6e565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fe280c6e92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fe280c6eaaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7fe280c6f921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7fe280c750ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7fe2840987e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7fe28409b468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7fe28409b90e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7fe283fd30cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7fe283f659fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7fe283f687ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7fe283f69250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7fe2806045b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fe280604668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..6fe1ac7 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f10d0c85565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f10d0c8592b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f10d0c85aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f10d0c86921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f10d0c8c0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f10d40af7e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f10d40b2468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f10d40b290e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f10d3fea0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f10d3f7c9fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f10d3f7f7ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f10d3f80250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f10d061b5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f10d061b668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..6d43308 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f942c895565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f942c89592b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f942c895aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f942c896921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f942c89c0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f942fcbf7e6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f942fcc2468] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f942fcc290e] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f942fbfa0cb] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f942fb8c9fa] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f942fb8f7ae] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f942fb90250] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f942c22b5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f942c22b668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..14a8e33 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7feba5627565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7feba562792b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7feba5627aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7feba5628921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7feba562e0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7feba8de25a6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7feba8de5228] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7feba8de56ce] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7feba8d1d04b] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7feba8caf97a] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7feba8cb272e] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7feba8cb31d0] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7feba4fbd5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7feba4fbd668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..aca32d5 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f491b1d6565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f491b1d692b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f491b1d6aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f491b1d7921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f491b1dd0ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f491e9915a6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f491e994228] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f491e9946ce] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f491e8cc04b] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f491e85e97a] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f491e86172e] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f491e8621d0] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f491ab6c5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f491ab6c668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..224af50 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fd5e5ab1565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fd5e5ab192b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fd5e5ab1aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7fd5e5ab2921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7fd5e5ab80ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7fd5e926c5a6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7fd5e926f228] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7fd5e926f6ce] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7fd5e91a704b] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7fd5e913997a] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7fd5e913c72e] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7fd5e913d1d0] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7fd5e54475b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fd5e5447668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..4fee726 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f38f15cf565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f38f15cf92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f38f15cfaaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f38f15d0921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f38f15d60ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f38f4d8a5a6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f38f4d8d228] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f38f4d8d6ce] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f38f4cc504b] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f38f4c5797a] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f38f4c5a72e] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f38f4c5b1d0] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f38f0f655b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f38f0f65668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1-rocwmma] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__fa1__longctx16384__dual.log new file mode 100644 index 0000000..36a6672 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f565baf1565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f565baf192b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f565baf1aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f565baf2921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f565baf80ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f565f3835a6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f565f386228] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f565f3866ce] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f565f2be04b] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f565f25097a] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f565f25372e] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f565f2541d0] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f565b4875b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f565b487668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__fa1__longctx32768__dual.log new file mode 100644 index 0000000..516df01 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f11606b1565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f11606b192b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f11606b1aaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f11606b2921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f11606b80ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f1163f435a6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f1163f46228] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f1163f466ce] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f1163e7e04b] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f1163e1097a] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f1163e1372e] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f1163e141d0] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f11600475b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f1160047668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..67bb385 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fc19da7d565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fc19da7d92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fc19da7daaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7fc19da7e921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7fc19da840ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7fc1a130f5a6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7fc1a1312228] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7fc1a13126ce] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7fc1a124a04b] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7fc1a11dc97a] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7fc1a11df72e] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7fc1a11e01d0] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7fc19d4135b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fc19d413668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..76e9d00 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,25 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml.c:1679: GGML_ASSERT(obj_new) failed +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f03ea5fe565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f03ea5fe92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f03ea5feaaf] +/usr/local/lib64/libggml-base.so.0(+0x4921) [0x7f03ea5ff921] +/usr/local/lib64/libggml-base.so.0(ggml_view_4d+0x2e) [0x7f03ea6050ce] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next24build_delta_net_chunkingEP11ggml_tensorS1_S1_S1_S1_S1_S1_S1_i+0xf16) [0x7f03ede905a6] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3next23build_layer_attn_linearEP18llm_graph_input_rsP11ggml_tensorS3_S3_i+0xff8) [0x7f03ede93228] +/usr/local/lib64/libllama.so.0(_ZN19llm_build_qwen3nextC1ERK11llama_modelRK16llm_graph_params+0x16e) [0x7f03ede936ce] +/usr/local/lib64/libllama.so.0(_ZNK11llama_model11build_graphERK16llm_graph_params+0x90b) [0x7f03eddcb04b] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_reserveEjjjPK22llama_memory_context_ib+0xfa) [0x7f03edd5d97a] +/usr/local/lib64/libllama.so.0(_ZN13llama_contextC2ERK11llama_model20llama_context_params+0x103e) [0x7f03edd6072e] +/usr/local/lib64/libllama.so.0(llama_init_from_model+0x100) [0x7f03edd611d0] +/usr/local/bin/llama-bench() [0x4077bb] +/lib64/libc.so.6(+0x35b5) [0x7f03e9f945b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f03e9f94668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1] Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_amdvlk__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_amdvlk__fa1__longctx16384__dual.log new file mode 100644 index 0000000..7455ff7 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_amdvlk__fa1__longctx16384__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3next ?B Q4_K - Medium | 42.01 GiB | 79.67 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 487.22 ± 0.00 | +| qwen3next ?B Q4_K - Medium | 42.01 GiB | 79.67 B | Vulkan | 99 | 1 | tg32 @ d16384 | 40.26 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_amdvlk__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_amdvlk__fa1__longctx32768__dual.log new file mode 100644 index 0000000..b09dfaa --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_amdvlk__fa1__longctx32768__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3next ?B Q4_K - Medium | 42.01 GiB | 79.67 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 371.02 ± 0.00 | +| qwen3next ?B Q4_K - Medium | 42.01 GiB | 79.67 B | Vulkan | 99 | 1 | tg32 @ d32768 | 28.05 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_radv__fa1__longctx16384__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_radv__fa1__longctx16384__dual.log new file mode 100644 index 0000000..574b78f --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_radv__fa1__longctx16384__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3next ?B Q4_K - Medium | 42.01 GiB | 79.67 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 419.92 ± 0.00 | +| qwen3next ?B Q4_K - Medium | 42.01 GiB | 79.67 B | Vulkan | 99 | 1 | tg32 @ d16384 | 25.61 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_radv__fa1__longctx32768__dual.log b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_radv__fa1__longctx32768__dual.log new file mode 100644 index 0000000..bf762c5 --- /dev/null +++ b/benchmark/results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_radv__fa1__longctx32768__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| qwen3next ?B Q4_K - Medium | 42.01 GiB | 79.67 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 312.36 ± 0.00 | +| qwen3next ?B Q4_K - Medium | 42.01 GiB | 79.67 B | Vulkan | 99 | 1 | tg32 @ d32768 | 24.06 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..6fc313b --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1272.79 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.69 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..c61c56d --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1082.97 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.80 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..27ff473 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1065.38 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.68 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..d82fec1 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 947.18 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.81 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx16384__single.log new file mode 100644 index 0000000..a4b2631 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1047.11 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.90 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx32768__single.log new file mode 100644 index 0000000..1485c4e --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 834.79 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.89 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..9a03215 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 900.98 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.89 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..d17a1e5 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 741.81 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.87 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..98e9abb --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1289.14 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.65 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..102ef36 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 887.59 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.83 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..a8e3965 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1063.47 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.68 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..c4c10ff --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 776.30 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.81 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__fa1__longctx16384__single.log new file mode 100644 index 0000000..2625b26 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1549.39 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.89 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__fa1__longctx32768__single.log new file mode 100644 index 0000000..5efb85a --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1130.69 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.85 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..604ee87 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1239.05 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.87 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..dda956e --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 956.10 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.89 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..5965677 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1439.58 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.70 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..2215ada --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1025.11 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.82 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..ae61bc3 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1175.36 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.70 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..66a1b20 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 882.91 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.79 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__fa1__longctx16384__single.log new file mode 100644 index 0000000..baeb594 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1595.41 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.81 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__fa1__longctx32768__single.log new file mode 100644 index 0000000..6db8432 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1164.68 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.84 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..cf784f4 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1277.42 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.86 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..f80248a --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 995.61 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.84 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..4395536 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1283.45 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.69 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..c4bbaec --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 885.56 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.82 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..ec7d7ff --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1063.06 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.65 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..a61e39e --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 774.45 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.81 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__fa1__longctx16384__single.log new file mode 100644 index 0000000..1ae7a8b --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1537.91 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.87 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__fa1__longctx32768__single.log new file mode 100644 index 0000000..c43c165 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1125.61 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.88 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..444663f --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1233.44 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 28.86 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..e7b5bd6 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 954.85 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 26.81 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx16384__single.log new file mode 100644 index 0000000..1a7beb0 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 1204.67 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | Vulkan | 99 | 1 | tg32 @ d16384 | 28.45 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx32768__single.log new file mode 100644 index 0000000..892b679 --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 799.88 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | Vulkan | 99 | 1 | tg32 @ d32768 | 26.81 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_radv__fa1__longctx16384__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_radv__fa1__longctx16384__single.log new file mode 100644 index 0000000..4226d8b --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_radv__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 1034.34 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | Vulkan | 99 | 1 | tg32 @ d16384 | 21.76 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_radv__fa1__longctx32768__single.log b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_radv__fa1__longctx32768__single.log new file mode 100644 index 0000000..e92f8ca --- /dev/null +++ b/benchmark/results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_radv__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 724.25 ± 0.00 | +| gemma3 12B Q8_0 | 13.40 GiB | 11.77 B | Vulkan | 99 | 1 | tg32 @ d32768 | 19.48 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..8db8d11 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,24 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7ff1b9286565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7ff1b928692b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7ff1b9286aaf] +/usr/local/lib64/libggml-hip.so.0(+0x2cb5eb2) [0x7ff1bbff8eb2] +/usr/local/lib64/libggml-hip.so.0(+0x2cbafc9) [0x7ff1bbffdfc9] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x36d) [0x7ff1b92a0add] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7ff1bc6c2a90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7ff1bc6c4722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7ff1bc6c95df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7ff1bc6ca3fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7ff1b8c1c5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7ff1b8c1c668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..47b727b --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/gemma-3/BF16/gemma-3-27b-it-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..02e0bf5 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f0d6ba72565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f0d6ba7292b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f0d6ba72aaf] +/usr/local/lib64/libggml-hip.so.0(+0x2cb5eb2) [0x7f0d6e7e4eb2] +/usr/local/lib64/libggml-hip.so.0(+0x2cca317) [0x7f0d6e7f9317] +/usr/local/lib64/libggml-hip.so.0(+0x2cc892c) [0x7f0d6e7f792c] +/usr/local/lib64/libggml-hip.so.0(+0x2cc7af3) [0x7f0d6e7f6af3] +/usr/local/lib64/libggml-hip.so.0(+0x2cc258a) [0x7f0d6e7f158a] +/usr/local/lib64/libggml-hip.so.0(+0x2cbe8d4) [0x7f0d6e7ed8d4] +/usr/local/lib64/libggml-hip.so.0(+0x2cbb0ef) [0x7f0d6e7ea0ef] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7f0d6ba8cf63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7f0d6eeaea90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7f0d6eeb0722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7f0d6eeb55df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7f0d6eeb63fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7f0d6b4085b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f0d6b408668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..d3c0ea1 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/gemma-3/BF16/gemma-3-27b-it-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx16384__dual.log new file mode 100644 index 0000000..9bbd44d --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx16384__dual.log @@ -0,0 +1,24 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fa9d6d2f565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fa9d6d2f92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fa9d6d2faaf] +/usr/local/lib64/libggml-hip.so.0(+0x2da2e22) [0x7fa9d9b8ee22] +/usr/local/lib64/libggml-hip.so.0(+0x2da7f39) [0x7fa9d9b93f39] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x36d) [0x7fa9d6d49add] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7fa9da258a90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7fa9da25a722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7fa9da25f5df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7fa9da2603fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7fa9d66c55b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fa9d66c5668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx32768__dual.log new file mode 100644 index 0000000..2c96916 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/gemma-3/BF16/gemma-3-27b-it-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..ab21ff6 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f292267c565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f292267c92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f292267caaf] +/usr/local/lib64/libggml-hip.so.0(+0x2da2e22) [0x7f29254dbe22] +/usr/local/lib64/libggml-hip.so.0(+0x2db7287) [0x7f29254f0287] +/usr/local/lib64/libggml-hip.so.0(+0x2db589c) [0x7f29254ee89c] +/usr/local/lib64/libggml-hip.so.0(+0x2db4a63) [0x7f29254eda63] +/usr/local/lib64/libggml-hip.so.0(+0x2daf4fa) [0x7f29254e84fa] +/usr/local/lib64/libggml-hip.so.0(+0x2dab844) [0x7f29254e4844] +/usr/local/lib64/libggml-hip.so.0(+0x2da805f) [0x7f29254e105f] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7f2922696f63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7f2925ba5a90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7f2925ba7722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7f2925bac5df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7f2925bad3fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7f29220125b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f2922012668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7-nightly] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..c8f4320 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/gemma-3/BF16/gemma-3-27b-it-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..9a2f76e --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,24 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fa704bb9565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fa704bb992b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fa704bb9aaf] +/usr/local/lib64/libggml-hip.so.0(+0x2efaf92) [0x7fa707b70f92] +/usr/local/lib64/libggml-hip.so.0(+0x2f000e9) [0x7fa707b760e9] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x36d) [0x7fa704bd3add] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7fa70823ea90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7fa708240722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7fa7082455df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7fa7082463fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7fa70454f5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fa70454f668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7.9-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..524d315 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/gemma-3/BF16/gemma-3-27b-it-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..70f7b22 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f28a4d6b565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f28a4d6b92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f28a4d6baaf] +/usr/local/lib64/libggml-hip.so.0(+0x2efaf92) [0x7f28a7d22f92] +/usr/local/lib64/libggml-hip.so.0(+0x2f0f7f7) [0x7f28a7d377f7] +/usr/local/lib64/libggml-hip.so.0(+0x2f0ddcc) [0x7f28a7d35dcc] +/usr/local/lib64/libggml-hip.so.0(+0x2f0cf97) [0x7f28a7d34f97] +/usr/local/lib64/libggml-hip.so.0(+0x2f07a7b) [0x7f28a7d2fa7b] +/usr/local/lib64/libggml-hip.so.0(+0x2f03d4b) [0x7f28a7d2bd4b] +/usr/local/lib64/libggml-hip.so.0(+0x2f0020f) [0x7f28a7d2820f] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7f28a4d85f63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7f28a83f0a90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7f28a83f2722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7f28a83f75df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7f28a83f83fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7f28a47015b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f28a4701668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7.9-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..4285e3f --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/gemma-3/BF16/gemma-3-27b-it-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__fa1__longctx16384__dual.log new file mode 100644 index 0000000..30c3506 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +Hip error: 'out of memory'(2) at /therock/src/rocm-libraries/projects/hipblaslt/library/src/amd_detail/hipblaslt.cpp:147 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__fa1__longctx32768__dual.log new file mode 100644 index 0000000..818cf53 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/gemma-3/BF16/gemma-3-27b-it-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..7cb1809 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f0224efc565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f0224efc92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f0224efcaaf] +/usr/local/lib64/libggml-hip.so.0(+0x2fd1ef2) [0x7f0227f8aef2] +/usr/local/lib64/libggml-hip.so.0(+0x2fe6757) [0x7f0227f9f757] +/usr/local/lib64/libggml-hip.so.0(+0x2fe4d2c) [0x7f0227f9dd2c] +/usr/local/lib64/libggml-hip.so.0(+0x2fe3ef7) [0x7f0227f9cef7] +/usr/local/lib64/libggml-hip.so.0(+0x2fde9db) [0x7f0227f979db] +/usr/local/lib64/libggml-hip.so.0(+0x2fdacab) [0x7f0227f93cab] +/usr/local/lib64/libggml-hip.so.0(+0x2fd716f) [0x7f0227f9016f] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7f0224f16f63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7f0228658a90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7f022865a722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7f022865f5df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7f02286603fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7f02248925b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f0224892668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm-7.9] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..1a05dea --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to create context with model '/home/kyuz0/models/gemma-3/BF16/gemma-3-27b-it-BF16-00001-of-00002.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..5cd8cfb --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f08f345f565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f08f345f92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f08f345faaf] +/usr/local/lib64/libggml-hip.so.0(+0x1b3df2) [0x7f08f36cfdf2] +/usr/local/lib64/libggml-hip.so.0(+0x1c8677) [0x7f08f36e4677] +/usr/local/lib64/libggml-hip.so.0(+0x1c6c28) [0x7f08f36e2c28] +/usr/local/lib64/libggml-hip.so.0(+0x1c5dda) [0x7f08f36e1dda] +/usr/local/lib64/libggml-hip.so.0(+0x1c081a) [0x7f08f36dc81a] +/usr/local/lib64/libggml-hip.so.0(+0x1bcae6) [0x7f08f36d8ae6] +/usr/local/lib64/libggml-hip.so.0(+0x1b905f) [0x7f08f36d505f] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7f08f3479f63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7f08f66cfa90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7f08f66d1722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7f08f66d65df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7f08f66d73fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7f08f2df55b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f08f2df5668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..6094e36 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f52a0362565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f52a036292b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f52a0362aaf] +/usr/local/lib64/libggml-hip.so.0(+0x1b3df2) [0x7f52a05d2df2] +/usr/local/lib64/libggml-hip.so.0(+0x1c8677) [0x7f52a05e7677] +/usr/local/lib64/libggml-hip.so.0(+0x1c6b78) [0x7f52a05e5b78] +/usr/local/lib64/libggml-hip.so.0(+0x1c5dda) [0x7f52a05e4dda] +/usr/local/lib64/libggml-hip.so.0(+0x1c081a) [0x7f52a05df81a] +/usr/local/lib64/libggml-hip.so.0(+0x1bcae6) [0x7f52a05dbae6] +/usr/local/lib64/libggml-hip.so.0(+0x1b905f) [0x7f52a05d805f] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7f52a037cf63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7f52a35d2a90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7f52a35d4722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7f52a35d95df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7f52a35da3fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7f529fcf85b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f529fcf8668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..fedd641 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f9fa3d7b565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f9fa3d7b92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f9fa3d7baaf] +/usr/local/lib64/libggml-hip.so.0(+0x1b3df2) [0x7f9fa3febdf2] +/usr/local/lib64/libggml-hip.so.0(+0x1c8677) [0x7f9fa4000677] +/usr/local/lib64/libggml-hip.so.0(+0x1c6c28) [0x7f9fa3ffec28] +/usr/local/lib64/libggml-hip.so.0(+0x1c5dda) [0x7f9fa3ffddda] +/usr/local/lib64/libggml-hip.so.0(+0x1c081a) [0x7f9fa3ff881a] +/usr/local/lib64/libggml-hip.so.0(+0x1bcae6) [0x7f9fa3ff4ae6] +/usr/local/lib64/libggml-hip.so.0(+0x1b905f) [0x7f9fa3ff105f] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7f9fa3d95f63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7f9fa6feba90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7f9fa6fed722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7f9fa6ff25df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7f9fa6ff33fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7f9fa37115b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f9fa3711668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..b145768 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fd5d04fc565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fd5d04fc92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fd5d04fcaaf] +/usr/local/lib64/libggml-hip.so.0(+0x1b3df2) [0x7fd5d076cdf2] +/usr/local/lib64/libggml-hip.so.0(+0x1c8677) [0x7fd5d0781677] +/usr/local/lib64/libggml-hip.so.0(+0x1c6b78) [0x7fd5d077fb78] +/usr/local/lib64/libggml-hip.so.0(+0x1c5dda) [0x7fd5d077edda] +/usr/local/lib64/libggml-hip.so.0(+0x1c081a) [0x7fd5d077981a] +/usr/local/lib64/libggml-hip.so.0(+0x1bcae6) [0x7fd5d0775ae6] +/usr/local/lib64/libggml-hip.so.0(+0x1b905f) [0x7fd5d077205f] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7fd5d0516f63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7fd5d376ca90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7fd5d376e722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7fd5d37735df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7fd5d37743fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7fd5cfe925b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fd5cfe92668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__fa1__longctx16384__dual.log new file mode 100644 index 0000000..d0b4ebf --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__fa1__longctx16384__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fb5110ee565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fb5110ee92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fb5110eeaaf] +/usr/local/lib64/libggml-hip.so.0(+0x1b3d52) [0x7fb51135ed52] +/usr/local/lib64/libggml-hip.so.0(+0x1c85d7) [0x7fb5113735d7] +/usr/local/lib64/libggml-hip.so.0(+0x1c6b88) [0x7fb511371b88] +/usr/local/lib64/libggml-hip.so.0(+0x1c5d3a) [0x7fb511370d3a] +/usr/local/lib64/libggml-hip.so.0(+0x1c077a) [0x7fb51136b77a] +/usr/local/lib64/libggml-hip.so.0(+0x1bca46) [0x7fb511367a46] +/usr/local/lib64/libggml-hip.so.0(+0x1b8fbf) [0x7fb511363fbf] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7fb511108f63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7fb5143e3a90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7fb5143e5722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7fb5143ea5df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7fb5143eb3fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7fb510a845b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fb510a84668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__fa1__longctx32768__dual.log new file mode 100644 index 0000000..f8e28b8 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__fa1__longctx32768__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f873ac8f565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f873ac8f92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f873ac8faaf] +/usr/local/lib64/libggml-hip.so.0(+0x1b3d52) [0x7f873aeffd52] +/usr/local/lib64/libggml-hip.so.0(+0x1c85d7) [0x7f873af145d7] +/usr/local/lib64/libggml-hip.so.0(+0x1c6ad8) [0x7f873af12ad8] +/usr/local/lib64/libggml-hip.so.0(+0x1c5d3a) [0x7f873af11d3a] +/usr/local/lib64/libggml-hip.so.0(+0x1c077a) [0x7f873af0c77a] +/usr/local/lib64/libggml-hip.so.0(+0x1bca46) [0x7f873af08a46] +/usr/local/lib64/libggml-hip.so.0(+0x1b8fbf) [0x7f873af04fbf] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7f873aca9f63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7f873df84a90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7f873df86722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7f873df8b5df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7f873df8c3fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7f873a6255b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f873a625668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..c7cbae4 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f00a3359565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f00a335992b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f00a3359aaf] +/usr/local/lib64/libggml-hip.so.0(+0x1b3d52) [0x7f00a35c9d52] +/usr/local/lib64/libggml-hip.so.0(+0x1c85d7) [0x7f00a35de5d7] +/usr/local/lib64/libggml-hip.so.0(+0x1c6b88) [0x7f00a35dcb88] +/usr/local/lib64/libggml-hip.so.0(+0x1c5d3a) [0x7f00a35dbd3a] +/usr/local/lib64/libggml-hip.so.0(+0x1c077a) [0x7f00a35d677a] +/usr/local/lib64/libggml-hip.so.0(+0x1bca46) [0x7f00a35d2a46] +/usr/local/lib64/libggml-hip.so.0(+0x1b8fbf) [0x7f00a35cefbf] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7f00a3373f63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7f00a664ea90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7f00a6650722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7f00a66555df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7f00a66563fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7f00a2cef5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f00a2cef668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..77d4ed8 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fb2c2652565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fb2c265292b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fb2c2652aaf] +/usr/local/lib64/libggml-hip.so.0(+0x1b3d52) [0x7fb2c28c2d52] +/usr/local/lib64/libggml-hip.so.0(+0x1c85d7) [0x7fb2c28d75d7] +/usr/local/lib64/libggml-hip.so.0(+0x1c6ad8) [0x7fb2c28d5ad8] +/usr/local/lib64/libggml-hip.so.0(+0x1c5d3a) [0x7fb2c28d4d3a] +/usr/local/lib64/libggml-hip.so.0(+0x1c077a) [0x7fb2c28cf77a] +/usr/local/lib64/libggml-hip.so.0(+0x1bca46) [0x7fb2c28cba46] +/usr/local/lib64/libggml-hip.so.0(+0x1b8fbf) [0x7fb2c28c7fbf] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7fb2c266cf63] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7fb2c5947a90] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7fb2c5949722] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7fb2c594e5df] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7fb2c594f3fe] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7fb2c1fe85b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fb2c1fe8668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm6_4_4] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..6c22b3a --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7f176c065565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7f176c06592b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7f176c065aaf] +/usr/local/lib64/libggml-hip.so.0(+0x2efbf92) [0x7f176f01df92] +/usr/local/lib64/libggml-hip.so.0(+0x2f107f7) [0x7f176f0327f7] +/usr/local/lib64/libggml-hip.so.0(+0x2f0edcc) [0x7f176f030dcc] +/usr/local/lib64/libggml-hip.so.0(+0x2f0df97) [0x7f176f02ff97] +/usr/local/lib64/libggml-hip.so.0(+0x2f08a7b) [0x7f176f02aa7b] +/usr/local/lib64/libggml-hip.so.0(+0x2f04d4b) [0x7f176f026d4b] +/usr/local/lib64/libggml-hip.so.0(+0x2f0120f) [0x7f176f02320f] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7f176c07ffe3] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7f176f6eba10] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7f176f6ed6a2] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7f176f6f255f] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7f176f6f337e] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7f176b9fb5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7f176b9fb668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..08b9914 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,6 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +✖ ! [rocm7.1.1-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..8cb147e --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fa331a7f565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fa331a7f92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fa331a7faaf] +/usr/local/lib64/libggml-hip.so.0(+0x2efbf92) [0x7fa334a37f92] +/usr/local/lib64/libggml-hip.so.0(+0x2f107f7) [0x7fa334a4c7f7] +/usr/local/lib64/libggml-hip.so.0(+0x2f0edcc) [0x7fa334a4adcc] +/usr/local/lib64/libggml-hip.so.0(+0x2f0df97) [0x7fa334a49f97] +/usr/local/lib64/libggml-hip.so.0(+0x2f08a7b) [0x7fa334a44a7b] +/usr/local/lib64/libggml-hip.so.0(+0x2f04d4b) [0x7fa334a40d4b] +/usr/local/lib64/libggml-hip.so.0(+0x2f0120f) [0x7fa334a3d20f] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7fa331a99fe3] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7fa335105a10] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7fa3351076a2] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7fa33510c55f] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7fa33510d37e] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7fa3314155b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fa331415668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..286d5eb --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,6 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +✖ ! [rocm7.1.1-rocwmma] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__fa1__longctx16384__dual.log new file mode 100644 index 0000000..8e15f9b --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__fa1__longctx16384__dual.log @@ -0,0 +1,24 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fa3b1dfc565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fa3b1dfc92b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fa3b1dfcaaf] +/usr/local/lib64/libggml-hip.so.0(+0x2fd2ef2) [0x7fa3b4e8bef2] +/usr/local/lib64/libggml-hip.so.0(+0x2fd8049) [0x7fa3b4e91049] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x36d) [0x7fa3b1e16b5d] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7fa3b5559a10] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7fa3b555b6a2] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7fa3b556055f] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7fa3b556137e] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7fa3b17925b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fa3b1792668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__fa1__longctx32768__dual.log new file mode 100644 index 0000000..fb28b25 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__fa1__longctx32768__dual.log @@ -0,0 +1,6 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +✖ ! [rocm7.1.1] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..869806d --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,29 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +/opt/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu:92: ROCm error +/usr/local/lib64/libggml-base.so.0(+0x3565) [0x7fcdb1288565] +/usr/local/lib64/libggml-base.so.0(ggml_print_backtrace+0x1eb) [0x7fcdb128892b] +/usr/local/lib64/libggml-base.so.0(ggml_abort+0x11f) [0x7fcdb1288aaf] +/usr/local/lib64/libggml-hip.so.0(+0x2fd2ef2) [0x7fcdb4317ef2] +/usr/local/lib64/libggml-hip.so.0(+0x2fe7757) [0x7fcdb432c757] +/usr/local/lib64/libggml-hip.so.0(+0x2fe5d2c) [0x7fcdb432ad2c] +/usr/local/lib64/libggml-hip.so.0(+0x2fe4ef7) [0x7fcdb4329ef7] +/usr/local/lib64/libggml-hip.so.0(+0x2fdf9db) [0x7fcdb43249db] +/usr/local/lib64/libggml-hip.so.0(+0x2fdbcab) [0x7fcdb4320cab] +/usr/local/lib64/libggml-hip.so.0(+0x2fd816f) [0x7fcdb431d16f] +/usr/local/lib64/libggml-base.so.0(ggml_backend_sched_graph_compute_async+0x7f3) [0x7fcdb12a2fe3] +/usr/local/lib64/libllama.so.0(_ZN13llama_context13graph_computeEP11ggml_cgraphb+0xa0) [0x7fcdb49e5a10] +/usr/local/lib64/libllama.so.0(_ZN13llama_context14process_ubatchERK12llama_ubatch14llm_graph_typeP22llama_memory_context_iR11ggml_status+0xe2) [0x7fcdb49e76a2] +/usr/local/lib64/libllama.so.0(_ZN13llama_context6decodeERK11llama_batch+0x3bf) [0x7fcdb49ec55f] +/usr/local/lib64/libllama.so.0(llama_decode+0xe) [0x7fcdb49ed37e] +/usr/local/bin/llama-bench() [0x40acdb] +/usr/local/bin/llama-bench() [0x4087ec] +/lib64/libc.so.6(+0x35b5) [0x7fcdb0c1e5b5] +/lib64/libc.so.6(__libc_start_main+0x88) [0x7fcdb0c1e668] +/usr/local/bin/llama-bench() [0x409b65] +✖ ! [rocm7.1.1] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..700dea3 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,6 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +✖ ! [rocm7.1.1] gemma-3-27b-it-BF16-00001-of-00002__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx16384__dual.log new file mode 100644 index 0000000..e50c824 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx16384__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +ggml_vulkan: Device memory allocation of size 2819260416 failed. +ggml_vulkan: Requested buffer size exceeds device buffer size limit: ErrorOutOfDeviceMemory +main: error: failed to load model '/home/kyuz0/models/gemma-3/BF16/gemma-3-27b-it-BF16-00001-of-00002.gguf' +✖ ! [vulkan_amdvlk] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx32768__dual.log new file mode 100644 index 0000000..b2a788e --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx32768__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +ggml_vulkan: Device memory allocation of size 2819260416 failed. +ggml_vulkan: Requested buffer size exceeds device buffer size limit: ErrorOutOfDeviceMemory +main: error: failed to load model '/home/kyuz0/models/gemma-3/BF16/gemma-3-27b-it-BF16-00001-of-00002.gguf' +✖ ! [vulkan_amdvlk] gemma-3-27b-it-BF16-00001-of-00002__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_radv__fa1__longctx16384__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_radv__fa1__longctx16384__dual.log new file mode 100644 index 0000000..38cffc0 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_radv__fa1__longctx16384__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gemma3 27B BF16 | 50.31 GiB | 27.01 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 648.18 ± 0.00 | +| gemma3 27B BF16 | 50.31 GiB | 27.01 B | Vulkan | 99 | 1 | tg32 @ d16384 | 8.64 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_radv__fa1__longctx32768__dual.log b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_radv__fa1__longctx32768__dual.log new file mode 100644 index 0000000..8acd346 --- /dev/null +++ b/benchmark/results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_radv__fa1__longctx32768__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gemma3 27B BF16 | 50.31 GiB | 27.01 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 541.45 ± 0.00 | +| gemma3 27B BF16 | 50.31 GiB | 27.01 B | Vulkan | 99 | 1 | tg32 @ d32768 | 8.31 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..70f4b80 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 4759.11 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 109.45 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..8660fb7 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 3642.83 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 101.43 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..01071b9 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 4763.38 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 109.46 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..cbb65af --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 3742.66 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 101.21 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__fa1__longctx16384__single.log new file mode 100644 index 0000000..a5d56b9 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3677.16 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 113.73 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__fa1__longctx32768__single.log new file mode 100644 index 0000000..d6da5fb --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2691.11 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 103.21 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..e02a85b --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3693.31 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 113.84 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..25499a6 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2692.18 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 103.40 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..c844328 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3213.82 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 108.42 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..f5a3e56 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2230.85 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 99.82 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..97d2aad --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3105.94 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 108.13 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..c882cfa --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2235.18 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 99.77 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__fa1__longctx16384__single.log new file mode 100644 index 0000000..39ba48d --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3651.25 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 111.88 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__fa1__longctx32768__single.log new file mode 100644 index 0000000..e75eb04 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2668.70 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 101.94 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..d30a560 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3669.78 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 112.11 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..a2455ed --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm-7.9__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2669.82 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 101.62 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..7fa6956 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3536.78 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 108.74 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..074f554 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2503.00 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 100.43 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..9ea5dc9 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3553.14 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 108.76 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..476664b --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2532.36 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 100.52 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__fa1__longctx16384__single.log new file mode 100644 index 0000000..23162a6 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3702.74 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 112.89 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__fa1__longctx32768__single.log new file mode 100644 index 0000000..f7fe49f --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2739.96 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 102.41 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..2c4fa41 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3716.52 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 112.78 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..9c2c101 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2725.02 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 102.67 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..7242dae --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3159.92 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 107.62 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..bf3c249 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2231.91 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 99.90 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..1eac8fa --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3192.77 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 107.90 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..9db978e --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2240.78 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 99.78 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__fa1__longctx16384__single.log new file mode 100644 index 0000000..2a6b79e --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3648.62 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 112.01 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__fa1__longctx32768__single.log new file mode 100644 index 0000000..5994f31 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2658.30 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 102.01 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__hblt0__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..33ae84e --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 3642.88 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 111.94 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__hblt0__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..0c69ece --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 2668.72 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 101.77 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_amdvlk__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_amdvlk__fa1__longctx16384__single.log new file mode 100644 index 0000000..ef001f5 --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_amdvlk__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 2716.15 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | Vulkan | 99 | 1 | tg32 @ d16384 | 101.87 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_amdvlk__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_amdvlk__fa1__longctx32768__single.log new file mode 100644 index 0000000..66e6d4f --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_amdvlk__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 2025.77 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | Vulkan | 99 | 1 | tg32 @ d32768 | 100.14 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_radv__fa1__longctx16384__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_radv__fa1__longctx16384__single.log new file mode 100644 index 0000000..ccc1f2e --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_radv__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 2326.04 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | Vulkan | 99 | 1 | tg32 @ d16384 | 85.50 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_radv__fa1__longctx32768__single.log b/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_radv__fa1__longctx32768__single.log new file mode 100644 index 0000000..142e13f --- /dev/null +++ b/benchmark/results/gemma-3-4b-it-Q3_K_S__vulkan_radv__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 1650.23 ± 0.00 | +| gemma3 4B Q3_K - Small | 1.80 GiB | 3.88 B | Vulkan | 99 | 1 | tg32 @ d32768 | 63.36 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..4b83ddb --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..fe887ab --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..571581e --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..ecf8dc2 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__fa1__longctx16384__dual.log new file mode 100644 index 0000000..5832595 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__fa1__longctx32768__dual.log new file mode 100644 index 0000000..23152ac --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..e42e811 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..8432522 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7-nightly] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..89e9695 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..4167a6d --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..16a8741 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..3262db5 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__fa1__longctx16384__dual.log new file mode 100644 index 0000000..db7c6ea --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__fa1__longctx32768__dual.log new file mode 100644 index 0000000..2d914fe --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..0a2d00c --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..339b175 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm-7.9] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..0d07f36 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..5e33c84 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..22968a3 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..2f80200 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__fa1__longctx16384__dual.log new file mode 100644 index 0000000..abb939e --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__fa1__longctx32768__dual.log new file mode 100644 index 0000000..19a5340 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..ad7de8d --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..f8ab36c --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm6_4_4] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log new file mode 100644 index 0000000..09e7b55 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log new file mode 100644 index 0000000..21b6b26 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..8e6e9e1 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..b08bccf --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1-rocwmma] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__fa1__longctx16384__dual.log new file mode 100644 index 0000000..841d45b --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__fa1__longctx32768__dual.log new file mode 100644 index 0000000..524e885 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1] gpt-oss-120b-mxfp4-00001-of-00003__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__hblt0__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__hblt0__fa1__longctx16384__dual.log new file mode 100644 index 0000000..c8fb823 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__hblt0__fa1__longctx16384__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx16384 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__hblt0__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__hblt0__fa1__longctx32768__dual.log new file mode 100644 index 0000000..8f8e469 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__hblt0__fa1__longctx32768__dual.log @@ -0,0 +1,9 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 2 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 + Device 1: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +main: error: failed to load model '/home/kyuz0/models/gpt-oss-120b/gpt-oss-120b-mxfp4-00001-of-00003.gguf' +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +✖ ! [rocm7.1.1] gpt-oss-120b-mxfp4-00001-of-00003__hblt0__fa1 __longctx32768 failed (exit 0) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_amdvlk__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_amdvlk__fa1__longctx16384__dual.log new file mode 100644 index 0000000..4cdfc94 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_amdvlk__fa1__longctx16384__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gpt-oss 120B MXFP4 MoE | 59.02 GiB | 116.83 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 863.78 ± 0.00 | +| gpt-oss 120B MXFP4 MoE | 59.02 GiB | 116.83 B | Vulkan | 99 | 1 | tg32 @ d16384 | 53.14 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_amdvlk__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_amdvlk__fa1__longctx32768__dual.log new file mode 100644 index 0000000..a9fd4dc --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_amdvlk__fa1__longctx32768__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gpt-oss 120B MXFP4 MoE | 59.02 GiB | 116.83 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 516.47 ± 0.00 | +| gpt-oss 120B MXFP4 MoE | 59.02 GiB | 116.83 B | Vulkan | 99 | 1 | tg32 @ d32768 | 45.01 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_radv__fa1__longctx16384__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_radv__fa1__longctx16384__dual.log new file mode 100644 index 0000000..9460848 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_radv__fa1__longctx16384__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gpt-oss 120B MXFP4 MoE | 59.02 GiB | 116.83 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 913.28 ± 0.00 | +| gpt-oss 120B MXFP4 MoE | 59.02 GiB | 116.83 B | Vulkan | 99 | 1 | tg32 @ d16384 | 59.93 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_radv__fa1__longctx32768__dual.log b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_radv__fa1__longctx32768__dual.log new file mode 100644 index 0000000..9b61b57 --- /dev/null +++ b/benchmark/results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_radv__fa1__longctx32768__dual.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gpt-oss 120B MXFP4 MoE | 59.02 GiB | 116.83 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 608.15 ± 0.00 | +| gpt-oss 120B MXFP4 MoE | 59.02 GiB | 116.83 B | Vulkan | 99 | 1 | tg32 @ d32768 | 24.30 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..a00cf85 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1559.64 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 110.83 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..91c8251 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 965.98 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 94.05 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..510405d --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1565.69 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 110.65 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..adeb88e --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 989.33 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 94.04 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__fa1__longctx16384__single.log new file mode 100644 index 0000000..c076173 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1432.01 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 116.48 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__fa1__longctx32768__single.log new file mode 100644 index 0000000..187b925 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 931.89 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 103.84 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__hblt0__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..d257f93 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1434.90 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 116.39 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__hblt0__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..83c509b --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7-nightly__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 930.75 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 103.92 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..66c0e31 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1856.32 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 108.91 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..20df0bd --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1080.53 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 92.54 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..c4da0b3 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1861.03 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 108.82 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..35d1ac2 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1079.67 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 92.66 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__fa1__longctx16384__single.log new file mode 100644 index 0000000..82e3ff7 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1704.10 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 113.69 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__fa1__longctx32768__single.log new file mode 100644 index 0000000..c657d72 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1033.61 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 101.47 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__hblt0__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..751afcb --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1708.49 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 113.27 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__hblt0__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..9c9c544 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm-7.9__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1033.50 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 101.42 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..f13c9bd --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 2075.85 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 108.78 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..b93f0d7 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1228.38 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 92.61 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..c1b908b --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 2069.60 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 108.78 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..8600c7e --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1236.65 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 92.54 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__fa1__longctx16384__single.log new file mode 100644 index 0000000..34a4fda --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 2036.77 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 119.50 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__fa1__longctx32768__single.log new file mode 100644 index 0000000..b648501 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1243.44 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 109.45 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__hblt0__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..79bf5a4 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 2036.55 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 120.23 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__hblt0__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..ab39de3 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm6_4_4__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1256.54 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 109.29 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..5ade9c3 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1874.38 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 108.75 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..12436fc --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1083.90 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 92.69 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..3b891f9 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1870.43 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 108.93 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..967cec9 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1091.55 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 92.77 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__fa1__longctx16384__single.log new file mode 100644 index 0000000..39a09f2 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1701.41 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 113.29 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__fa1__longctx32768__single.log new file mode 100644 index 0000000..e8731e1 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 983.49 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 101.52 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__hblt0__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..a6536a2 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1707.25 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 113.38 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__hblt0__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..2178aa2 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__rocm7.1.1__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 1024.54 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 101.23 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__vulkan_amdvlk__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__vulkan_amdvlk__fa1__longctx16384__single.log new file mode 100644 index 0000000..4aeceda --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__vulkan_amdvlk__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 1572.29 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | Vulkan | 99 | 1 | tg32 @ d16384 | 93.73 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__vulkan_amdvlk__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__vulkan_amdvlk__fa1__longctx32768__single.log new file mode 100644 index 0000000..51b62e7 --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__vulkan_amdvlk__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 831.30 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | Vulkan | 99 | 1 | tg32 @ d32768 | 71.03 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__vulkan_radv__fa1__longctx16384__single.log b/benchmark/results/gpt-oss-20b-mxfp4__vulkan_radv__fa1__longctx16384__single.log new file mode 100644 index 0000000..52deaef --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__vulkan_radv__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 1677.77 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | Vulkan | 99 | 1 | tg32 @ d16384 | 97.97 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/gpt-oss-20b-mxfp4__vulkan_radv__fa1__longctx32768__single.log b/benchmark/results/gpt-oss-20b-mxfp4__vulkan_radv__fa1__longctx32768__single.log new file mode 100644 index 0000000..c54d33c --- /dev/null +++ b/benchmark/results/gpt-oss-20b-mxfp4__vulkan_radv__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 1011.88 ± 0.00 | +| gpt-oss 20B MXFP4 MoE | 11.27 GiB | 20.91 B | Vulkan | 99 | 1 | tg32 @ d32768 | 81.88 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..1195c94 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1026.89 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.04 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..4fea7ca --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 644.96 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 23.70 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..67ffb66 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1027.68 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.12 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..c4dbaac --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 645.86 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 23.67 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__fa1__longctx16384__single.log new file mode 100644 index 0000000..881a577 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 846.26 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.11 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__fa1__longctx32768__single.log new file mode 100644 index 0000000..3855284 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 520.67 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 23.65 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__hblt0__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..276aa0f --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 850.14 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 37.97 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__hblt0__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..eb5456b --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7-nightly__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 521.93 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 23.66 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..983e136 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 870.61 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.35 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..4d12ed4 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 487.39 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.01 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..0a2c1cd --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 870.44 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.30 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..e1cee33 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 486.62 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 23.99 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__fa1__longctx16384__single.log new file mode 100644 index 0000000..9186f73 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 983.17 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.44 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__fa1__longctx32768__single.log new file mode 100644 index 0000000..b7cf870 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 565.30 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.00 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__hblt0__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..cacbf7e --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 993.51 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.41 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__hblt0__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..fa02fc0 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm-7.9__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon Graphics, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 568.34 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.02 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..74beb14 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 915.22 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.47 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..fddb248 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 504.72 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.05 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..63b9b89 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 913.34 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.48 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..a633cf0 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 503.02 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.03 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__fa1__longctx16384__single.log new file mode 100644 index 0000000..579e54e --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1093.90 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.49 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__fa1__longctx32768__single.log new file mode 100644 index 0000000..4fe7161 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 628.44 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.04 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__hblt0__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..35b8e53 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 1096.44 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.48 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__hblt0__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..06cc1a3 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm6_4_4__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 631.38 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.03 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__fa1__longctx16384__single.log new file mode 100644 index 0000000..eb3281f --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 870.43 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.44 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__fa1__longctx32768__single.log new file mode 100644 index 0000000..ccb6e79 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 486.49 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.02 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..4f1192e --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 870.04 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.43 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..3a5580f --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 485.70 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 23.98 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__fa1__longctx16384__single.log new file mode 100644 index 0000000..2e2e54e --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 989.11 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.36 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__fa1__longctx32768__single.log new file mode 100644 index 0000000..bc3fa7c --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 566.70 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 23.97 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__hblt0__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__hblt0__fa1__longctx16384__single.log new file mode 100644 index 0000000..763d882 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__hblt0__fa1__longctx16384__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d16384 | 992.79 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d16384 | 38.38 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__hblt0__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__hblt0__fa1__longctx32768__single.log new file mode 100644 index 0000000..5ef0c96 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__rocm7.1.1__hblt0__fa1__longctx32768__single.log @@ -0,0 +1,10 @@ +ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no +ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no +ggml_cuda_init: found 1 ROCm devices: + Device 0: AMD Radeon AI PRO R9700, gfx1201 (0x1201), VMM: no, Wave Size: 32 +| model | size | params | backend | ngl | n_ubatch | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -------: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | pp2048 @ d32768 | 569.23 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | ROCm | 99 | 2048 | 1 | tg32 @ d32768 | 24.02 ± 0.00 | + +build: 22577583a (7312) diff --git a/benchmark/results/llama-2-7b.Q4_0__vulkan_amdvlk__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__vulkan_amdvlk__fa1__longctx16384__single.log new file mode 100644 index 0000000..4f938a9 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__vulkan_amdvlk__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 539.07 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | Vulkan | 99 | 1 | tg32 @ d16384 | 33.40 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__vulkan_amdvlk__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__vulkan_amdvlk__fa1__longctx32768__single.log new file mode 100644 index 0000000..f99d0c5 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__vulkan_amdvlk__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (AMD open-source driver) | uma: 0 | fp16: 1 | bf16: 0 | warp size: 64 | shared memory: 32768 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Ryzen 9 9900X3D 12-Core Processor (AMD open-source driver) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 32768 | int dot: 1 | matrix cores: none +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 292.34 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | Vulkan | 99 | 1 | tg32 @ d32768 | 18.08 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__vulkan_radv__fa1__longctx16384__single.log b/benchmark/results/llama-2-7b.Q4_0__vulkan_radv__fa1__longctx16384__single.log new file mode 100644 index 0000000..b547df9 --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__vulkan_radv__fa1__longctx16384__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | Vulkan | 99 | 1 | pp2048 @ d16384 | 808.11 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | Vulkan | 99 | 1 | tg32 @ d16384 | 38.64 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/benchmark/results/llama-2-7b.Q4_0__vulkan_radv__fa1__longctx32768__single.log b/benchmark/results/llama-2-7b.Q4_0__vulkan_radv__fa1__longctx32768__single.log new file mode 100644 index 0000000..ba0db7f --- /dev/null +++ b/benchmark/results/llama-2-7b.Q4_0__vulkan_radv__fa1__longctx32768__single.log @@ -0,0 +1,12 @@ +WARNING: radv is not a conformant Vulkan implementation, testing use only. +WARNING: radv is not a conformant Vulkan implementation, testing use only. +ggml_vulkan: Found 3 Vulkan devices: +ggml_vulkan: 0 = AMD Ryzen 9 9900X3D 12-Core Processor (RADV RAPHAEL_MENDOCINO) (radv) | uma: 1 | fp16: 1 | bf16: 0 | warp size: 32 | shared memory: 65536 | int dot: 1 | matrix cores: none +ggml_vulkan: 1 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +ggml_vulkan: 2 = AMD Radeon AI PRO R9700 (RADV GFX1201) (radv) | uma: 0 | fp16: 1 | bf16: 1 | warp size: 64 | shared memory: 65536 | int dot: 1 | matrix cores: KHR_coopmat +| model | size | params | backend | ngl | fa | test | t/s | +| ------------------------------ | ---------: | ---------: | ---------- | --: | -: | --------------: | -------------------: | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | Vulkan | 99 | 1 | pp2048 @ d32768 | 448.69 ± 0.00 | +| llama 7B Q4_0 | 3.56 GiB | 6.74 B | Vulkan | 99 | 1 | tg32 @ d32768 | 23.37 ± 0.00 | + +build: 6016d0bd4 (7285) diff --git a/docs/results.json b/docs/results.json index 4166571..621e7a1 100644 --- a/docs/results.json +++ b/docs/results.json @@ -1,6 +1,6 @@ { "meta": { - "generated_at": "2025-12-07T10:12:28Z", + "generated_at": "2025-12-07T16:55:28Z", "os_kernel": "Fedora 42 \u2014 Linux 6.15.9-201.fc42.x86_64 (Sat Aug 2 11:37:34 UTC 2025)", "llamacpp_builds": [ { @@ -41,6 +41,90 @@ "notes": "pp512 = prompt processing; tg128 = text generation; t/s = tokens/second" }, "runs": [ + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 959.74, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 17.59, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -67,6 +151,90 @@ "gpu_config": "single", "build": null }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 305.03, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 17.59, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -125,6 +293,90 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 761.86, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 18.02, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -183,6 +435,90 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 280.13, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 18.01, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -241,6 +577,90 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 665.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 17.95, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -299,6 +719,90 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 262.83, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 17.95, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -357,6 +861,90 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 767.43, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 18.07, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -415,6 +1003,90 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 275.13, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 18.07, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -473,6 +1145,90 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 688.43, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 17.97, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -531,6 +1287,90 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 271.52, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 17.98, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -589,6 +1429,90 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 829.03, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 18.1, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -647,6 +1571,90 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 290.54, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 18.11, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -905,6 +1913,90 @@ "number": "7312" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 665.8, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 17.95, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -963,6 +2055,90 @@ "number": "7312" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 262.76, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 17.95, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -1021,6 +2197,90 @@ "number": "7312" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 766.84, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 18.09, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -1079,6 +2339,90 @@ "number": "7312" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 274.75, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 18.07, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 14.0, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": null + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -1253,6 +2597,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 310.96, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 13.87, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 194.73, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 10.53, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -1311,6 +2771,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 550.05, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 15.14, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 351.12, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-14B-Instruct-2512-BF16", + "model_clean": "Ministral-3-14B-Instruct-2512-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 13.32, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 13.51, + "file_size_gib": 25.16, + "name_params_b": 13.51, + "quant": "BF16", + "log": "results/Ministral-3-14B-Instruct-2512-BF16__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-14B-Instruct-2512-BF16", "model_clean": "Ministral-3-14B-Instruct-2512-BF16", @@ -1369,6 +2945,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1210.41, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 26.73, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 708.36, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 22.94, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -1427,6 +3119,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 461.89, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 26.74, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 363.06, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 22.91, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -1485,6 +3293,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 958.03, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.59, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 552.93, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.8, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -1543,6 +3467,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 413.39, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.58, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 313.0, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.79, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -1601,6 +3641,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 832.67, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.47, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 459.52, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -1659,6 +3815,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 384.05, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.46, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 282.23, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.46, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -1717,6 +3989,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 963.98, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.76, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 552.83, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.91, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -1775,6 +4163,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 406.44, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.71, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 308.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.9, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -1833,6 +4337,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 868.42, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.51, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 494.21, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.48, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -1891,6 +4511,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 397.69, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.47, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 295.42, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.47, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -1949,6 +4685,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1058.69, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 621.76, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.94, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -2007,6 +4859,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 433.9, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 336.61, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.93, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -2297,6 +5265,122 @@ "number": "7312" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 832.53, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.47, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 463.8, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.48, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -2355,6 +5439,122 @@ "number": "7312" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 383.9, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.46, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 281.71, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -2413,6 +5613,122 @@ "number": "7312" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 964.51, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.76, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 544.22, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.91, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -2471,6 +5787,122 @@ "number": "7312" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 406.04, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 27.77, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 308.59, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.92, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -2645,6 +6077,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 403.34, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 18.97, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 242.85, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 14.1, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -2703,6 +6251,122 @@ "number": "7285" } }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 718.51, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 22.17, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 442.1, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Ministral-3-8B-Instruct-2512-BF16", + "model_clean": "Ministral-3-8B-Instruct-2512-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 18.81, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 8.49, + "file_size_gib": 15.81, + "name_params_b": 8.49, + "quant": "BF16", + "log": "results/Ministral-3-8B-Instruct-2512-BF16__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Ministral-3-8B-Instruct-2512-BF16", "model_clean": "Ministral-3-8B-Instruct-2512-BF16", @@ -2819,6 +6483,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -2877,6 +6593,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -2935,6 +6703,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -2993,6 +6813,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3051,6 +6923,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3109,6 +7033,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3167,6 +7143,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3225,6 +7253,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3283,6 +7363,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3341,6 +7473,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3399,6 +7583,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3457,6 +7693,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3747,6 +8035,58 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3805,6 +8145,58 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3863,6 +8255,58 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -3921,6 +8365,58 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 30.0, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -4063,6 +8559,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 158.35, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 56.89, + "name_params_b": 30.53, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 12.09, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 56.89, + "name_params_b": 30.53, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 108.88, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 56.89, + "name_params_b": 30.53, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 12.02, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 56.89, + "name_params_b": 30.53, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", @@ -4121,6 +8733,238 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 594.9, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 56.89, + "name_params_b": 30.53, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_radv__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 13.86, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 56.89, + "name_params_b": 30.53, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_radv__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 285.47, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 56.89, + "name_params_b": 30.53, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_radv__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 9.77, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 56.89, + "name_params_b": 30.53, + "quant": "BF16", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-BF16-00001-of-00002__vulkan_radv__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1009.52, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 69.94, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 556.12, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 50.17, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -4179,6 +9023,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1009.56, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 69.86, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 559.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 51.14, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -4237,6 +9197,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 745.53, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 78.82, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 408.05, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 63.42, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -4295,6 +9371,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 743.47, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 78.71, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 409.2, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 63.48, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -4353,6 +9545,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 624.67, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 72.28, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 336.79, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 56.71, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -4411,6 +9719,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 625.56, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 72.41, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 341.42, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 53.72, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -4469,6 +9893,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 730.98, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 73.7, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 397.89, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 59.19, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -4527,6 +10067,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 736.16, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 73.3, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 406.18, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 59.22, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -4585,6 +10241,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 668.97, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 73.51, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 364.36, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 57.28, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -4643,6 +10415,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 669.72, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 73.74, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 366.16, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 56.94, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -4701,6 +10589,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 837.05, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 75.93, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 471.22, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 60.49, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -4759,6 +10763,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 844.53, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 76.5, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 477.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 61.07, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -5049,6 +11169,122 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 622.46, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 72.32, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 341.42, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 56.64, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -5107,6 +11343,122 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 623.52, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 71.99, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 341.6, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 56.65, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -5165,6 +11517,122 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 733.54, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 73.55, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 395.85, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 59.13, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -5223,6 +11691,122 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 736.46, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 73.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 406.36, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 59.16, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -5397,6 +11981,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 389.88, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 49.62, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 201.27, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 23.92, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -5455,6 +12155,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 591.92, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 64.3, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 332.93, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 46.15, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 17.35, + "name_params_b": 30.53, + "quant": "Q4_K_M", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-Q4_K_M__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M", @@ -5571,6 +12387,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 925.95, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 56.08, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 496.65, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 43.58, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -5629,6 +12561,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 880.38, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 56.2, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 518.19, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 43.57, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -5687,6 +12735,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 700.17, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 61.73, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 323.72, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 52.03, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -5745,6 +12909,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 674.08, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 61.71, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 387.83, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 52.17, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -5803,6 +13083,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 629.39, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 59.02, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 342.35, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 49.03, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -5861,6 +13257,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 608.61, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 59.46, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 335.81, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 49.03, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -5919,6 +13431,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 737.67, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 59.93, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 407.9, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 50.8, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -5977,6 +13605,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 711.88, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 59.7, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 399.61, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 50.76, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -6035,6 +13779,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 669.98, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 59.94, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 368.96, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 49.4, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -6093,6 +13953,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 645.77, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 59.74, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 360.83, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 49.26, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -6151,6 +14127,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 846.1, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 61.46, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 477.88, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 52.14, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -6209,6 +14301,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 814.93, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 61.38, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 466.97, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 51.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -6499,6 +14707,122 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 625.6, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 59.35, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 342.65, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 48.97, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -6557,6 +14881,122 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 608.95, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 59.43, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 336.21, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 49.08, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -6615,6 +15055,122 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 742.91, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 59.63, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 408.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 50.72, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -6673,6 +15229,122 @@ "number": "7312" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 713.08, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 59.8, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 400.11, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 48.06, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -6847,6 +15519,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 403.05, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 41.38, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 207.11, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 25.15, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", @@ -6905,6 +15693,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 602.7, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_radv__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 53.51, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_radv__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 347.37, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_radv__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "model_clean": "Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 41.56, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 30.53, + "file_size_gib": 33.51, + "name_params_b": 30.53, + "quant": "Q8_K_XL", + "log": "results/Qwen3-Coder-30B-A3B-Instruct-UD-Q8_K_XL__vulkan_radv__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -6963,6 +15867,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7021,6 +15977,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7079,6 +16087,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7137,6 +16197,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7195,6 +16307,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7253,6 +16417,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7311,6 +16527,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7369,6 +16637,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm-7.9__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7427,6 +16747,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7485,6 +16857,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7543,6 +16967,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7601,6 +17077,58 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7891,6 +17419,58 @@ "number": "7312" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -7949,6 +17529,58 @@ "number": "7312" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -8007,6 +17639,58 @@ "number": "7312" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -8065,6 +17749,58 @@ "number": "7312" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": 80.0, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -8239,6 +17975,122 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 487.22, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 79.67, + "file_size_gib": 42.01, + "name_params_b": 79.67, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_amdvlk__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 40.26, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 79.67, + "file_size_gib": 42.01, + "name_params_b": 79.67, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_amdvlk__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 371.02, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 79.67, + "file_size_gib": 42.01, + "name_params_b": 79.67, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_amdvlk__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 28.05, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 79.67, + "file_size_gib": 42.01, + "name_params_b": 79.67, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_amdvlk__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", @@ -8297,6 +18149,238 @@ "number": "7285" } }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 419.92, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 79.67, + "file_size_gib": 42.01, + "name_params_b": 79.67, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_radv__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 25.61, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 79.67, + "file_size_gib": 42.01, + "name_params_b": 79.67, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_radv__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 312.36, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 79.67, + "file_size_gib": 42.01, + "name_params_b": 79.67, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_radv__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "model_clean": "Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.06, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 79.67, + "file_size_gib": 42.01, + "name_params_b": 79.67, + "quant": "Q4_K_XL", + "log": "results/Qwen3-Next-80B-A3B-Instruct-UD-Q4_K_XL__vulkan_radv__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1272.79, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.69, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1082.97, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.8, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -8323,6 +18407,122 @@ "gpu_config": "single", "build": null }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1065.38, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.68, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 947.18, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.81, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -8381,6 +18581,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1047.11, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.9, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 834.79, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.89, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -8439,6 +18755,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 900.98, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.89, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 741.81, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.87, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -8497,6 +18929,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1289.14, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.65, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 887.59, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.83, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -8555,6 +19103,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1063.47, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.68, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 776.3, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.81, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -8613,6 +19277,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1549.39, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.89, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1130.69, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.85, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -8671,6 +19451,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1239.05, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.87, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 956.1, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.89, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -8729,6 +19625,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1439.58, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.7, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1025.11, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.82, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -8787,6 +19799,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1175.36, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.7, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 882.91, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.79, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -8845,6 +19973,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1595.41, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.81, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1164.68, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.84, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -8903,6 +20147,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1277.42, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.86, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 995.61, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.84, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -9193,6 +20553,122 @@ "number": "7312" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1283.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.69, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 885.56, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.82, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -9251,6 +20727,122 @@ "number": "7312" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1063.06, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.65, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 774.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.81, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -9309,6 +20901,122 @@ "number": "7312" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1537.91, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.87, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1125.61, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.88, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -9367,6 +21075,122 @@ "number": "7312" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1233.44, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.86, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 954.85, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.81, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -9541,6 +21365,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1204.67, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 28.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 799.88, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 26.81, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -9599,6 +21539,122 @@ "number": "7285" } }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1034.34, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 21.76, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 724.25, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-12b-it-UD-Q8_K_XL", + "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 19.48, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 11.77, + "file_size_gib": 13.4, + "name_params_b": 11.77, + "quant": "Q8_K_XL", + "log": "results/gemma-3-12b-it-UD-Q8_K_XL__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-12b-it-UD-Q8_K_XL", "model_clean": "gemma-3-12b-it-UD-Q8_K_XL", @@ -9715,6 +21771,58 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -9773,6 +21881,58 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -9831,6 +21991,58 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -9889,6 +22101,58 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -9915,6 +22179,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -9973,6 +22289,58 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10031,6 +22399,58 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10089,6 +22509,58 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm-7.9__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10147,6 +22619,58 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10205,6 +22729,58 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10263,6 +22839,58 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10321,6 +22949,58 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm6_4_4__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10579,6 +23259,58 @@ "number": "7312" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10637,6 +23369,58 @@ "number": "7312" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10695,6 +23479,58 @@ "number": "7312" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10753,6 +23589,58 @@ "number": "7312" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "runtime", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__rocm7.1.1__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10895,6 +23783,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_amdvlk__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gemma-3-27b-it-BF16-00001-of-00002", "model_clean": "gemma-3-27b-it-BF16", @@ -10953,6 +23893,238 @@ "number": "7285" } }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 648.18, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 27.01, + "file_size_gib": 50.31, + "name_params_b": 27.01, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_radv__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 8.64, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 27.01, + "file_size_gib": 50.31, + "name_params_b": 27.01, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_radv__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 541.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 27.01, + "file_size_gib": 50.31, + "name_params_b": 27.01, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_radv__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-27b-it-BF16-00001-of-00002", + "model_clean": "gemma-3-27b-it-BF16", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 8.31, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 27.01, + "file_size_gib": 50.31, + "name_params_b": 27.01, + "quant": "BF16", + "log": "results/gemma-3-27b-it-BF16-00001-of-00002__vulkan_radv__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 4759.11, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 109.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 3642.83, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 101.43, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11011,6 +24183,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 4763.38, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 109.46, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 3742.66, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 101.21, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11069,6 +24357,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3677.16, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 113.73, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2691.11, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 103.21, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11127,6 +24531,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3693.31, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 113.84, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2692.18, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 103.4, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11185,6 +24705,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3213.82, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 108.42, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2230.85, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 99.82, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11243,6 +24879,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3105.94, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 108.13, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2235.18, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 99.77, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11301,6 +25053,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3651.25, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 111.88, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2668.7, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 101.94, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11359,6 +25227,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3669.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 112.11, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2669.82, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 101.62, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11417,6 +25401,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3536.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 108.74, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2503.0, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 100.43, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11475,6 +25575,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3553.14, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 108.76, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2532.36, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 100.52, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11533,6 +25749,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3702.74, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 112.89, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2739.96, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 102.41, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11591,6 +25923,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3716.52, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 112.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2725.02, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 102.67, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11881,6 +26329,122 @@ "number": "7312" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3159.92, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 107.62, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2231.91, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 99.9, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11939,6 +26503,122 @@ "number": "7312" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3192.77, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 107.9, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2240.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 99.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -11997,6 +26677,122 @@ "number": "7312" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3648.62, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 112.01, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2658.3, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 102.01, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -12055,6 +26851,122 @@ "number": "7312" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 3642.88, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 111.94, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2668.72, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 101.77, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -12229,6 +27141,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 2716.15, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 101.87, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 2025.77, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 100.14, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -12287,6 +27315,122 @@ "number": "7285" } }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 2326.04, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 85.5, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1650.23, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gemma-3-4b-it-Q3_K_S", + "model_clean": "gemma-3-4b-it-Q3_K_S", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 63.36, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 3.88, + "file_size_gib": 1.8, + "name_params_b": 3.88, + "quant": "Q3_K_S", + "log": "results/gemma-3-4b-it-Q3_K_S__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gemma-3-4b-it-Q3_K_S", "model_clean": "gemma-3-4b-it-Q3_K_S", @@ -12371,6 +27515,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12397,6 +27593,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12423,6 +27671,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12449,6 +27749,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7-nightly__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12475,6 +27827,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12501,6 +27905,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12527,6 +27983,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12553,6 +28061,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm-7.9__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12579,6 +28139,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12605,6 +28217,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12631,6 +28295,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12657,6 +28373,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm6_4_4__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12787,6 +28555,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12813,6 +28633,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12839,6 +28711,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12865,6 +28789,58 @@ "gpu_config": "dual", "build": null }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__hblt0__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": null, + "tps_mean": null, + "tps_std": null, + "error": true, + "error_type": "load", + "backend": null, + "ngl": null, + "mmap": null, + "params_b": null, + "file_size_gib": null, + "name_params_b": null, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__rocm7.1.1__hblt0__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": null + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -12975,6 +28951,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 863.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 116.83, + "file_size_gib": 59.02, + "name_params_b": 116.83, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_amdvlk__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 53.14, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 116.83, + "file_size_gib": 59.02, + "name_params_b": 116.83, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_amdvlk__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 516.47, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 116.83, + "file_size_gib": 59.02, + "name_params_b": 116.83, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_amdvlk__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 45.01, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 116.83, + "file_size_gib": 59.02, + "name_params_b": 116.83, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_amdvlk__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-120b-mxfp4-00001-of-00003", "model_clean": "gpt-oss-120b-mxfp4", @@ -13033,6 +29125,238 @@ "number": "7285" } }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 913.28, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 116.83, + "file_size_gib": 59.02, + "name_params_b": 116.83, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_radv__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 59.93, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 116.83, + "file_size_gib": 59.02, + "name_params_b": 116.83, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_radv__fa1__longctx16384__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 608.15, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 116.83, + "file_size_gib": 59.02, + "name_params_b": 116.83, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_radv__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-120b-mxfp4-00001-of-00003", + "model_clean": "gpt-oss-120b-mxfp4", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.3, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 116.83, + "file_size_gib": 59.02, + "name_params_b": 116.83, + "quant": "MXFP4", + "log": "results/gpt-oss-120b-mxfp4-00001-of-00003__vulkan_radv__fa1__longctx32768__dual.log", + "rpc": false, + "gpu_config": "dual", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1559.64, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 110.83, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 965.98, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 94.05, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13091,6 +29415,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1565.69, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 110.65, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 989.33, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 94.04, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13149,6 +29589,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1432.01, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 116.48, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 931.89, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 103.84, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13207,6 +29763,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1434.9, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 116.39, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 930.75, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 103.92, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13265,6 +29937,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1856.32, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 108.91, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1080.53, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 92.54, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13323,6 +30111,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1861.03, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 108.82, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1079.67, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 92.66, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13381,6 +30285,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1704.1, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 113.69, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1033.61, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 101.47, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13439,6 +30459,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1708.49, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 113.27, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1033.5, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 101.42, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13497,6 +30633,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 2075.85, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 108.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1228.38, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 92.61, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13555,6 +30807,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 2069.6, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 108.78, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1236.65, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 92.54, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13613,6 +30981,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 2036.77, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 119.5, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1243.44, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 109.45, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13671,6 +31155,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 2036.55, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 120.23, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1256.54, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 109.29, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -13961,6 +31561,122 @@ "number": "7312" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1874.38, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 108.75, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1083.9, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 92.69, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -14019,6 +31735,122 @@ "number": "7312" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1870.43, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 108.93, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1091.55, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 92.77, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -14077,6 +31909,122 @@ "number": "7312" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1701.41, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 113.29, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 983.49, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 101.52, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -14135,6 +32083,122 @@ "number": "7312" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1707.25, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 113.38, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1024.54, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 101.23, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -14309,6 +32373,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1572.29, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 93.73, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 831.3, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 71.03, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -14367,6 +32547,122 @@ "number": "7285" } }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1677.77, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 97.97, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 1011.88, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "gpt-oss-20b-mxfp4", + "model_clean": "gpt-oss-20b-mxfp4", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 81.88, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 20.91, + "file_size_gib": 11.27, + "name_params_b": 20.91, + "quant": "MXFP4", + "log": "results/gpt-oss-20b-mxfp4__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "gpt-oss-20b-mxfp4", "model_clean": "gpt-oss-20b-mxfp4", @@ -14425,6 +32721,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1026.89, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.04, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 644.96, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-rocwmma", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 23.7, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -14483,6 +32895,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1027.68, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.12, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 645.86, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 23.67, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -14541,6 +33069,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 846.26, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.11, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 520.67, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly", + "env_base": "rocm", + "env_variant": "7-nightly", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 23.65, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -14599,6 +33243,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 850.14, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 37.97, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 521.93, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7-nightly-hblt0", + "env_base": "rocm", + "env_variant": "7-nightly-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 23.66, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7-nightly__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -14657,6 +33417,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 870.61, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.35, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 487.39, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-rocwmma", + "env_base": "rocm", + "env_variant": "7.9-rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.01, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -14715,6 +33591,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 870.44, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.3, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 486.62, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-rocwmma-hblt0", + "env_base": "rocm", + "env_variant": "7.9-rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 23.99, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -14773,6 +33765,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 983.17, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.44, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 565.3, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9", + "env_base": "rocm", + "env_variant": "7.9", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.0, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -14831,6 +33939,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 993.51, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.41, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 568.34, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm-7.9-hblt0", + "env_base": "rocm", + "env_variant": "7.9-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.02, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm-7.9__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -14889,6 +34113,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 915.22, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.47, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 504.72, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-rocwmma", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.05, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -14947,6 +34287,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 913.34, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.48, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 503.02, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-rocwmma-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.03, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -15005,6 +34461,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1093.9, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.49, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 628.44, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4", + "env_base": "rocm6_4_4", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.04, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -15063,6 +34635,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 1096.44, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.48, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 631.38, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm6_4_4-hblt0", + "env_base": "rocm6_4_4", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.03, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm6_4_4__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -15353,6 +35041,122 @@ "number": "7312" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 870.43, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.44, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 486.49, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-rocwmma", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.02, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -15411,6 +35215,122 @@ "number": "7312" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 870.04, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.43, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 485.7, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-rocwmma-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "rocwmma-hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 23.98, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1-rocwmma__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -15469,6 +35389,122 @@ "number": "7312" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 989.11, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.36, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 566.7, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1", + "env_base": "rocm7.1.1", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 23.97, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -15527,6 +35563,122 @@ "number": "7312" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 992.79, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.38, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1__hblt0__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 569.23, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "rocm7.1.1-hblt0", + "env_base": "rocm7.1.1", + "env_variant": "hblt0", + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 24.02, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "ROCm", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__rocm7.1.1__hblt0__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "22577583a", + "number": "7312" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -15701,6 +35853,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 539.07, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 33.4, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__vulkan_amdvlk__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 292.34, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "vulkan_amdvlk", + "env_base": "vulkan_amdvlk", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 18.08, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__vulkan_amdvlk__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0", @@ -15759,6 +36027,122 @@ "number": "7285" } }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "pp2048 @ d16384", + "tps_mean": 808.11, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx16384", + "context_tokens": 16384, + "test": "tg32 @ d16384", + "tps_mean": 38.64, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__vulkan_radv__fa1__longctx16384__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "pp2048 @ d32768", + "tps_mean": 448.69, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, + { + "model": "llama-2-7b.Q4_0", + "model_clean": "llama-2-7b.Q4_0", + "env": "vulkan_radv", + "env_base": "vulkan_radv", + "env_variant": null, + "fa": true, + "context": "longctx32768", + "context_tokens": 32768, + "test": "tg32 @ d32768", + "tps_mean": 23.37, + "tps_std": 0.0, + "error": false, + "error_type": null, + "backend": "Vulkan", + "ngl": 99, + "mmap": null, + "params_b": 6.74, + "file_size_gib": 3.56, + "name_params_b": 6.74, + "quant": "Q4_0", + "log": "results/llama-2-7b.Q4_0__vulkan_radv__fa1__longctx32768__single.log", + "rpc": false, + "gpu_config": "single", + "build": { + "hash": "6016d0bd4", + "number": "7285" + } + }, { "model": "llama-2-7b.Q4_0", "model_clean": "llama-2-7b.Q4_0",