cuda : fix multi-seq, quantized FA

ggml-ci
2025-11-06 09:46:50 +00:00 · 2025-07-22 20:48:53 +03:00
parent a856a5665d
commit 55cf48de1e
2 changed files with 16 additions and 6 deletions
--- a/tests/test-backend-ops.cpp
+++ b/tests/test-backend-ops.cpp
@@ -5525,6 +5525,8 @@ static std::vector<std::unique_ptr<test_case>> make_test_cases_eval() {
    test_cases.emplace_back(new test_timestep_embedding());
    test_cases.emplace_back(new test_leaky_relu());

+    test_cases.emplace_back(new test_flash_attn_ext(128, 128, 4, {1, 3}, 512, 128, true, 0.0f, 0.0f, GGML_PREC_DEFAULT, GGML_TYPE_Q8_0));
+
    for (int hsk : { 64, 80, 128, 192, 256, 576 }) {
        for (int hsv : { 64, 80, 128, 192, 256, 512 }) {
            if (hsk != 192 && hsk != 576 && hsk != hsv) continue;