LM/Qwen 3.8 27B: Difference between revisions

From Fundamental Ramen
< LM
Jump to navigation Jump to search
Line 68: Line 68:


<syntaxhighlight lang="markdown">
<syntaxhighlight lang="markdown">
./llama.cpp/sycl/bin/llama-bench -hf bartowski/Qwen3.8-27B-GGUF:IQ3_XXS -fa 0,1
./llama.cpp/sycl/bin/llama-bench -hf unsloth/Qwen3.8-27B-GGUF:UD-Q5_K_M -fa 0,1
 
| model                          |      size |    params | backend    | threads |  fa |            test |                  t/s |
| ------------------------------ | ---------: | ---------: | ---------- | ------: | --: | --------------: | -------------------: |
| qwen35 27B IQ3_XXS - 3.0625 bpw |  11.75 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          pp512 |        445.32 ± 0.53 |
| qwen35 27B IQ3_XXS - 3.0625 bpw |  11.75 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          tg128 |        10.28 ± 0.01 |
| qwen35 27B IQ3_XXS - 3.0625 bpw |  11.75 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          pp512 |        436.31 ± 0.21 |
| qwen35 27B IQ3_XXS - 3.0625 bpw |  11.75 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          tg128 |        10.23 ± 0.03 |
 
build: d7a207411 (10644)
 
./llama.cpp/sycl/bin/llama-bench -hf bartowski/Qwen3.8-27B-GGUF:Q4_0 -fa 0,1
 
| model                          |      size |    params | backend    | threads |  fa |            test |                  t/s |
| ------------------------------ | ---------: | ---------: | ---------- | ------: | --: | --------------: | -------------------: |
| qwen35 27B Q4_0                |  15.22 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          pp512 |        812.57 ± 2.08 |
| qwen35 27B Q4_0                |  15.22 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          tg128 |        26.89 ± 0.03 |
| qwen35 27B Q4_0                |  15.22 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          pp512 |        354.67 ± 1.02 |
| qwen35 27B Q4_0                |  15.22 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          tg128 |        26.77 ± 0.02 |
 
build: d7a207411 (10644)
 
./llama.cpp/sycl/bin/llama-bench -hf bartowski/Qwen3.8-27B-GGUF:Q4_K_L -fa 0,1
 
| model                          |      size |    params | backend    | threads |  fa |            test |                  t/s |
| ------------------------------ | ---------: | ---------: | ---------- | ------: | --: | --------------: | -------------------: |
| qwen35 27B Q4_K - Medium      |  17.42 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          pp512 |        867.21 ± 1.71 |
| qwen35 27B Q4_K - Medium      |  17.42 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          tg128 |        24.97 ± 0.09 |
| qwen35 27B Q4_K - Medium      |  17.42 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          pp512 |      756.78 ± 38.06 |
| qwen35 27B Q4_K - Medium      |  17.42 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          tg128 |        24.86 ± 0.09 |
 
build: d7a207411 (10644)
 
./llama.cpp/sycl/bin/llama-bench -hf bartowski/Qwen3.8-27B-GGUF:Q5_K_L -fa 0,1
 
| model                          |      size |    params | backend    | threads |  fa |            test |                  t/s |
| ------------------------------ | ---------: | ---------: | ---------- | ------: | --: | --------------: | -------------------: |
| qwen35 27B Q5_K - Medium      |  20.05 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          pp512 |      1031.14 ± 4.23 |
| qwen35 27B Q5_K - Medium      |  20.05 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          tg128 |        22.24 ± 0.03 |
| qwen35 27B Q5_K - Medium      |  20.05 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          pp512 |        833.60 ± 0.72 |
| qwen35 27B Q5_K - Medium      |  20.05 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          tg128 |        22.11 ± 0.03 |
 
build: d7a207411 (10644)
 
./llama.cpp/sycl/bin/llama-bench -hf bartowski/Qwen3.8-27B-GGUF:Q6_K_L -fa 0,1
 
| model                          |      size |    params | backend    | threads |  fa |            test |                  t/s |
| ------------------------------ | ---------: | ---------: | ---------- | ------: | --: | --------------: | -------------------: |
| qwen35 27B Q6_K                |  22.42 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          pp512 |        984.56 ± 5.09 |
| qwen35 27B Q6_K                |  22.42 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          tg128 |        20.08 ± 0.03 |
| qwen35 27B Q6_K                |  22.42 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          pp512 |        715.19 ± 1.40 |
| qwen35 27B Q6_K                |  22.42 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          tg128 |        19.93 ± 0.09 |
 
build: d7a207411 (10644)
 
./llama.cpp/sycl/bin/llama-bench -hf unsloth/Qwen3.8-27B-GGUF:Q4_0 -fa 0,1
 
| model                          |      size |    params | backend    | threads |  fa |            test |                  t/s |
| ------------------------------ | ---------: | ---------: | ---------- | ------: | --: | --------------: | -------------------: |
| qwen35 27B Q4_0                |  14.94 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          pp512 |        815.63 ± 1.66 |
| qwen35 27B Q4_0                |  14.94 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          tg128 |        27.59 ± 0.03 |
| qwen35 27B Q4_0                |  14.94 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          pp512 |        358.96 ± 1.03 |
| qwen35 27B Q4_0                |  14.94 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          tg128 |        27.40 ± 0.03 |
 
build: d7a207411 (10644)
 
./llama.cpp/sycl/bin/llama-bench -hf unsloth/Qwen3.8-27B-GGUF:UD-Q4_K_XL -fa 0,1
 
| model                          |      size |    params | backend    | threads |  fa |            test |                  t/s |
| ------------------------------ | ---------: | ---------: | ---------- | ------: | --: | --------------: | -------------------: |
| qwen35 27B Q4_K - Medium      |  16.34 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          pp512 |      727.84 ± 20.79 |
| qwen35 27B Q4_K - Medium      |  16.34 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          tg128 |        21.38 ± 0.05 |
| qwen35 27B Q4_K - Medium      |  16.34 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          pp512 |        754.42 ± 6.23 |
| qwen35 27B Q4_K - Medium      |  16.34 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          tg128 |        21.32 ± 0.05 |
 
build: d7a207411 (10644)
 
./llama.cpp/sycl/bin/llama-bench -hf unsloth/Qwen3.8-27B-GGUF:UD-Q5_K_XL -fa 0,1
./llama.cpp/sycl/bin/llama-bench -hf unsloth/Qwen3.8-27B-GGUF:UD-Q5_K_XL -fa 0,1
| model                          |      size |    params | backend    | threads |  fa |            test |                  t/s |
| ------------------------------ | ---------: | ---------: | ---------- | ------: | --: | --------------: | -------------------: |
| qwen35 27B Q5_K - Medium      |  19.43 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          pp512 |        978.10 ± 6.44 |
| qwen35 27B Q5_K - Medium      |  19.43 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          tg128 |        22.31 ± 0.03 |
| qwen35 27B Q5_K - Medium      |  19.43 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          pp512 |        843.67 ± 2.55 |
| qwen35 27B Q5_K - Medium      |  19.43 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          tg128 |        22.24 ± 0.02 |
build: d7a207411 (10644)
./llama.cpp/sycl/bin/llama-bench -hf unsloth/Qwen3.8-27B-GGUF:UD-Q6_K -fa 0,1
./llama.cpp/sycl/bin/llama-bench -hf unsloth/Qwen3.8-27B-GGUF:UD-Q6_K -fa 0,1
| model                          |      size |    params | backend    | threads |  fa |            test |                  t/s |
| ------------------------------ | ---------: | ---------: | ---------- | ------: | --: | --------------: | -------------------: |
| qwen35 27B Q6_K                |  20.46 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          pp512 |        973.08 ± 5.22 |
| qwen35 27B Q6_K                |  20.46 GiB |    27.32 B | SYCL,BLAS  |      6 |  0 |          tg128 |        21.21 ± 0.02 |
| qwen35 27B Q6_K                |  20.46 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          pp512 |        789.89 ± 2.14 |
| qwen35 27B Q6_K                |  20.46 GiB |    27.32 B | SYCL,BLAS  |      6 |  1 |          tg128 |        21.15 ± 0.02 |
build: d7a207411 (10644)
</syntaxhighlight>
</syntaxhighlight>



Revision as of 04:41, 31 August 2026

llama.cpp

Running

# works
./llama.cpp/sycl/bin/llama-server \
  -hf unsloth/Qwen3.8-27B-GGUF:UD-Q6_K_M \
  --spec-type draft-mtp --spec-draft-n-max 2 -ngld 99 \
  -c 64000 -ctk q8_0 -ctv q8_0 \
  -fa on -ngl 99 --load-mode mlock -np 1 --jinja --reasoning-preserve -t 8

# with turboquant
./llama.cpp/vulkan/bin/llama-server \
  -hf unsloth/Qwen3.8-27B-GGUF:UD-Q6_K_M \
  --spec-type draft-mtp --spec-draft-n-max 2 -ngld 99 \
  -c 64000 -ctk q8_0 -ctv turbo4 \
  -fa on -ngl 99 --load-mode mlock -np 1 --jinja --reasoning-preserve -t 8

Benchmark for me

# control group
./llama.cpp/sycl/bin/llama-bench \
  -hf unsloth/Qwen3.8-27B-GGUF:UD-Q5_K_XL \
  -fa on -ctk q8_0 -ctv q8_0 -ngl 99

| model                          |       size |     params | backend    | ngl | type_k | type_v |  fa |            test |                  t/s |
| ------------------------------ | ---------: | ---------: | ---------- | --: | -----: | -----: | --: | --------------: | -------------------: |
| qwen35 27B Q5_K - Medium       |  19.43 GiB |    27.32 B | SYCL       |  99 |   q8_0 |   q8_0 |   1 |           pp512 |        951.05 ± 7.72 |
| qwen35 27B Q5_K - Medium       |  19.43 GiB |    27.32 B | SYCL       |  99 |   q8_0 |   q8_0 |   1 |           tg128 |         22.17 ± 0.06 |

# Test -ctv
./llama.cpp/sycl/bin/llama-bench \
  -hf unsloth/Qwen3.8-27B-GGUF:UD-Q5_K_XL \
  -fa on -ctk q8_0 -ctv q4_0 -ngl 99

# Test -ub
./llama.cpp/sycl/bin/llama-bench \
  -hf unsloth/Qwen3.8-27B-GGUF:UD-Q5_K_XL \
  -fa on -ctk q8_0 -ctv q8_0 -ngl 99 -ub 2048

# Test -fa off
./llama.cpp/sycl/bin/llama-bench \
  -hf unsloth/Qwen3.8-27B-GGUF:UD-Q5_K_XL \
  -fa off -ngl 99

# Test bartowski
./llama.cpp/sycl/bin/llama-bench \
  -hf bartowski/Qwen3.8-27B-GGUF:Q5_K_L \
  -fa on -ctk q8_0 -ctv q8_0 -ngl 99

| model                          |       size |     params | backend    | ngl | type_k | type_v |  fa |            test |                  t/s |
| ------------------------------ | ---------: | ---------: | ---------- | --: | -----: | -----: | --: | --------------: | -------------------: |
| qwen35 27B Q5_K - Medium       |  20.05 GiB |    27.32 B | SYCL       |  99 |   q8_0 |   q8_0 |   1 |           pp512 |        997.64 ± 2.90 |
| qwen35 27B Q5_K - Medium       |  20.05 GiB |    27.32 B | SYCL       |  99 |   q8_0 |   q8_0 |   1 |           tg128 |         22.08 ± 0.06 |

# Test OpenVINO
GGML_OPENVINO_DEVICE=GPU GGML_OPENVINO_STATEFUL_EXECUTION=1 ./llama.cpp/openvino/bin/llama-bench \
  -hf unsloth/Qwen3.8-27B-GGUF:Q4_0 \
  -fa on -ngl 99

Benchmark for community

./llama.cpp/sycl/bin/llama-bench -hf unsloth/Qwen3.8-27B-GGUF:UD-Q5_K_M -fa 0,1
./llama.cpp/sycl/bin/llama-bench -hf unsloth/Qwen3.8-27B-GGUF:UD-Q5_K_XL -fa 0,1
./llama.cpp/sycl/bin/llama-bench -hf unsloth/Qwen3.8-27B-GGUF:UD-Q6_K -fa 0,1

Experiment

# search

hf models ls --search "Qwen3.8-27B" --apps llama.cpp --expand "downloads,likes,lastModified" --sort downloads --no-truncate --limit 10

# unsloth

hf models ls --tree -R -h unsloth/Qwen3.8-27B-GGUF

hf download unsloth/Qwen3.8-27B-GGUF --include "*UD-IQ4_XS*"
hf download unsloth/Qwen3.8-27B-GGUF --include "*UD-Q4_K_XL*"
hf download unsloth/Qwen3.8-27B-GGUF --include "*UD-Q5_K_XL*"
hf download unsloth/Qwen3.8-27B-GGUF --include "*UD-Q6_K.gguf"
hf download unsloth/Qwen3.8-27B-GGUF --include "*UD-Q6_K_M*"
hf download unsloth/Qwen3.8-27B-GGUF --include "*UD-Q6_K_L*"
hf download unsloth/Qwen3.8-27B-GGUF --include "*UD-Q6_K_XL*"
hf download unsloth/Qwen3.8-27B-GGUF --include "*UD-Q8_K_L*"
hf download unsloth/Qwen3.8-27B-GGUF --include "*Q4_0*"
hf download unsloth/Qwen3.8-27B-GGUF --include "*Q8_0*"
hf download unsloth/Qwen3.8-27B-GGUF --include "mmproj-BF16*"
hf download unsloth/Qwen3.8-27B-GGUF --include "*mtp-Qwen3.8-27B-Q4_0.gguf"

find ~/.cache/huggingface/hub -name "*Qwen3.8-27B-UD-Q8_K_L.gguf"
find ~/.cache/huggingface/hub -name "*Qwen3.8-27B-UD-Q8_K_L.gguf" -exec sh -c 'rm -f "$1" "$(readlink -f "$1")"' _ {} \;

tree -h ~/.cache/huggingface/hub/models--unsloth--Qwen3.8-27B-GGUF

# bartowski

hf models ls --tree -R -h bartowski/Qwen3.8-27B-GGUF

hf download bartowski/Qwen3.8-27B-GGUF --include "*IQ3_XXS*"
hf download bartowski/Qwen3.8-27B-GGUF --include "*IQ4_XS*"
hf download bartowski/Qwen3.8-27B-GGUF --include "*Q4_0*"
hf download bartowski/Qwen3.8-27B-GGUF --include "*Q4_1*"
hf download bartowski/Qwen3.8-27B-GGUF --include "*Q4_K_L*"
hf download bartowski/Qwen3.8-27B-GGUF --include "*Q5_K_L*"
hf download bartowski/Qwen3.8-27B-GGUF --include "*Q6_K_L*"
hf download bartowski/Qwen3.8-27B-GGUF --include "*Q8_0*"
hf download bartowski/Qwen3.8-27B-GGUF --include "*mmproj-Qwen3.8-27B-bf16.gguf*"

tree -h ~/.cache/huggingface/hub/models--bartowski--Qwen3.8-27B-GGUF

Accuracy

https://unsloth.ai/docs/~gitbook/image?url=https%3A%2F%2F3215535692-files.gitbook.io%2F%7E%2Ffiles%2Fv0%2Fb%2Fgitbook-x-prod.appspot.com%2Fo%2Fspaces%252FxhOjnexMCB3dmuQFQ2Zq%252Fuploads%252Fl9piThTmmUsePst5w3F2%252Fimage.png%3Falt%3Dmedia%26token%3D821e88e2-ad4e-40bf-adf2-e13837228e82&width=768&dpr=3&quality=100&sign=dd5f7cb5&sv=2