From dba1e27fa838e306d7cc6f9131f3a7792095b8d8 Mon Sep 17 00:00:00 2001 From: Daniel Hiltgen Date: Thu, 2 Jul 2026 17:11:39 -0700 Subject: [PATCH] llama: enable FA on CUDA CC 6.x GPUs (#16994) Recent upstream Pascal kernel fixes let us compile native SM60/SM61 kernels again instead of relying on PTX JIT, so allow Flash Attention auto at runtime for CC 6.x devices. Fixes #16591 Fixes #16754 --- docs/development.md | 4 ++-- docs/faq.mdx | 2 +- llama/server/CMakePresets.json | 2 +- llm/llama_server_test.go | 2 +- ml/device.go | 2 +- ml/device_test.go | 15 +++++++++++++-- 6 files changed, 19 insertions(+), 8 deletions(-) diff --git a/docs/development.md b/docs/development.md index ef28cee8..dec7b555 100644 --- a/docs/development.md +++ b/docs/development.md @@ -51,10 +51,10 @@ cmake -B build . -DOLLAMA_LLAMA_BACKENDS=cuda_v13 -DCMAKE_CUDA_ARCHITECTURES=nat cmake -B build . -DOLLAMA_LLAMA_BACKENDS=rocm_v7_2 -DCMAKE_HIP_ARCHITECTURES=gfx1100 ``` -You can tune GGML build options by setting `GGML_*` values during configure. For example, to build CUDA v12 for Pascal without flash attention kernels: +You can tune GGML build options by setting `GGML_*` values during configure. For example, to disable CUDA flash attention kernels for local debugging: ```shell -cmake -B build . -DOLLAMA_LLAMA_BACKENDS=cuda_v12 -DCMAKE_CUDA_ARCHITECTURES=61 -DGGML_CUDA_FA=OFF +cmake -B build . -DOLLAMA_LLAMA_BACKENDS=cuda_v12 -DGGML_CUDA_FA=OFF ``` ## macOS (Apple Silicon) diff --git a/docs/faq.mdx b/docs/faq.mdx index 47f06529..7cedacec 100644 --- a/docs/faq.mdx +++ b/docs/faq.mdx @@ -343,7 +343,7 @@ When loading a new model, Ollama evaluates the required VRAM for the model again ## How can I enable Flash Attention? -Flash Attention is a feature of most modern models that can significantly reduce memory usage as the context size grows. To enable Flash Attention, set the `OLLAMA_FLASH_ATTENTION` environment variable to `1` when starting the Ollama server. +Flash Attention is a feature of most modern models that can significantly reduce memory usage as the context size grows. Ollama uses Flash Attention automatically when the selected backend and devices support it. To force Flash Attention on, set `OLLAMA_FLASH_ATTENTION=1` when starting the Ollama server. To disable it, set `OLLAMA_FLASH_ATTENTION=0`. ## How can I set the quantization type for the K/V cache? diff --git a/llama/server/CMakePresets.json b/llama/server/CMakePresets.json index ddabece2..9ac2f6aa 100644 --- a/llama/server/CMakePresets.json +++ b/llama/server/CMakePresets.json @@ -68,7 +68,7 @@ "inherits": ["llama_cuda_v12_base"], "binaryDir": "${sourceDir}/../../build/llama-server-cuda_v12", "cacheVariables": { - "CMAKE_CUDA_ARCHITECTURES": "50-virtual;52-virtual;60-virtual;61-virtual;70;75;80;86;89;90;90a;120" + "CMAKE_CUDA_ARCHITECTURES": "50-virtual;52-virtual;60;61;70;75;80;86;89;90;90a;120" } }, { diff --git a/llm/llama_server_test.go b/llm/llama_server_test.go index 9e52c07c..70e66302 100644 --- a/llm/llama_server_test.go +++ b/llm/llama_server_test.go @@ -1955,7 +1955,7 @@ func TestAppendFlashAttentionArgs(t *testing.T) { supportedGPU := []ml.DeviceInfo{{DeviceID: ml.DeviceID{Library: "CUDA"}, DriverMajor: 13, ComputeMajor: 8, ComputeMinor: 9}} oldGPU := []ml.DeviceInfo{ {DeviceID: ml.DeviceID{Library: "CUDA"}, DriverMajor: 12, ComputeMajor: 8, ComputeMinor: 9}, - {DeviceID: ml.DeviceID{Library: "CUDA"}, DriverMajor: 12, ComputeMajor: 6, ComputeMinor: 2}, + {DeviceID: ml.DeviceID{Library: "CUDA"}, DriverMajor: 12, ComputeMajor: 5, ComputeMinor: 0}, } tests := []struct { diff --git a/ml/device.go b/ml/device.go index 5dee4fe6..fbeb0713 100644 --- a/ml/device.go +++ b/ml/device.go @@ -585,7 +585,7 @@ func FlashAttentionSupported(l []DeviceInfo) bool { func cudaFlashAttentionSupported(gpu DeviceInfo) bool { if gpu.Library != "CUDA" || - gpu.ComputeMajor < 7 || + gpu.ComputeMajor < 6 || (gpu.ComputeMajor == 7 && gpu.ComputeMinor == 2) { return false } diff --git a/ml/device_test.go b/ml/device_test.go index 1bb52573..474974f4 100644 --- a/ml/device_test.go +++ b/ml/device_test.go @@ -186,8 +186,19 @@ func TestFlashAttentionSupported(t *testing.T) { gpus: []DeviceInfo{{DeviceID: DeviceID{Library: "CUDA"}, DriverMajor: 12, ComputeMajor: 5, ComputeMinor: 0}}, }, { - name: "cuda compute 6.2 unsupported", + name: "cuda compute 6.0 supported", + gpus: []DeviceInfo{{DeviceID: DeviceID{Library: "CUDA"}, DriverMajor: 12, ComputeMajor: 6, ComputeMinor: 0}}, + want: true, + }, + { + name: "cuda compute 6.1 supported", + gpus: []DeviceInfo{{DeviceID: DeviceID{Library: "CUDA"}, DriverMajor: 12, ComputeMajor: 6, ComputeMinor: 1}}, + want: true, + }, + { + name: "cuda compute 6.2 supported", gpus: []DeviceInfo{{DeviceID: DeviceID{Library: "CUDA"}, DriverMajor: 12, ComputeMajor: 6, ComputeMinor: 2}}, + want: true, }, { name: "cuda compute 7.2 unsupported", @@ -216,7 +227,7 @@ func TestFlashAttentionSupported(t *testing.T) { name: "mixed cuda unsupported", gpus: []DeviceInfo{ {DeviceID: DeviceID{Library: "CUDA"}, DriverMajor: 12, ComputeMajor: 8, ComputeMinor: 9}, - {DeviceID: DeviceID{Library: "CUDA"}, DriverMajor: 12, ComputeMajor: 6, ComputeMinor: 2}, + {DeviceID: DeviceID{Library: "CUDA"}, DriverMajor: 12, ComputeMajor: 5, ComputeMinor: 0}, }, }, {