llm: introduce k/v context quantization (vRAM improvements) (#6279)

2025-12-09 23:37:06 +00:00 · 2024-12-04 10:57:19 +11:00
parent 2b82c5a8a1
commit 1bdab9fdb1
10 changed files with 147 additions and 21 deletions
--- a/discover/types.go
+++ b/discover/types.go
@@ -183,3 +183,17 @@ func (si SystemInfo) GetOptimalThreadCount() int {

 	return coreCount
 }
+
+// For each GPU, check if it does NOT support flash attention
+func (l GpuInfoList) FlashAttentionSupported() bool {
+	for _, gpu := range l {
+		supportsFA := gpu.Library == "metal" ||
+			(gpu.Library == "cuda" && gpu.DriverMajor >= 7) ||
+			gpu.Library == "rocm"
+
+		if !supportsFA {
+			return false
+		}
+	}
+	return true
+}