-
Notifications
You must be signed in to change notification settings - Fork 87
ggml-cpu: enable Q2_0 VNNI kernel on AVX-VNNI-only CPUs #76
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
base: prism
Are you sure you want to change the base?
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -568,9 +568,17 @@ void ggml_vec_dot_q2_0_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const voi | |
|
|
||
| float sumf = 0.0f; | ||
|
|
||
| #if defined(__AVX512VNNI__) && defined(__AVX512VL__) | ||
| // AVX-512-VNNI: unpack 2-bit codes c in {0,1,2,3} (value = c-1), then | ||
| #if (defined(__AVX512VNNI__) && defined(__AVX512VL__)) || defined(__AVXVNNI__) | ||
| // VNNI: unpack 2-bit codes c in {0,1,2,3} (value = c-1), then | ||
| // dot((c-1), qy) = dpbusd(c, qy) - dpbusd(1, qy). | ||
| // The kernel only uses 256-bit registers, so it runs unchanged on | ||
| // AVX-VNNI-only CPUs (e.g. Intel Alder/Raptor Lake, where AVX512 is | ||
| // unavailable); the AVX-VNNI intrinsic differs only in name. | ||
| #if defined(__AVX512VNNI__) && defined(__AVX512VL__) | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Consider dropping the macro and calling |
||
| #define GGML_Q2_0_DPBUSD(acc, a, b) _mm256_dpbusd_epi32(acc, a, b) | ||
| #else | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. If the macro stays, make this |
||
| #define GGML_Q2_0_DPBUSD(acc, a, b) _mm256_dpbusd_avx_epi32(acc, a, b) | ||
| #endif | ||
| const __m256i ones = _mm256_set1_epi8(1); | ||
| const __m128i idxlo = _mm_setr_epi8(0,0,0,0,1,1,1,1,2,2,2,2,3,3,3,3); | ||
| const __m128i idxhi = _mm_setr_epi8(4,4,4,4,5,5,5,5,6,6,6,6,7,7,7,7); | ||
|
|
@@ -591,12 +599,13 @@ void ggml_vec_dot_q2_0_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const voi | |
| r0 = _mm256_and_si256(_mm256_srli_epi16(_mm256_mullo_epi16(r0, mul), 6), three); | ||
| r1 = _mm256_and_si256(_mm256_srli_epi16(_mm256_mullo_epi16(r1, mul), 6), three); | ||
| __m256i codes = _mm256_permute4x64_epi64(_mm256_packus_epi16(r0, r1), 0xD8); // 32 codes in order | ||
| const int dp = hsum_i32_8(_mm256_dpbusd_epi32(_mm256_setzero_si256(), codes, qy)); | ||
| const int sy = hsum_i32_8(_mm256_dpbusd_epi32(_mm256_setzero_si256(), ones, qy)); | ||
| const int dp = hsum_i32_8(GGML_Q2_0_DPBUSD(_mm256_setzero_si256(), codes, qy)); | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. The two reductions can be one: keep the dpbusd results as vectors and do |
||
| const int sy = hsum_i32_8(GGML_Q2_0_DPBUSD(_mm256_setzero_si256(), ones, qy)); | ||
| sumi += d1 * (float)(dp - sy); | ||
| } | ||
| sumf += d0 * sumi; | ||
| } | ||
| #undef GGML_Q2_0_DPBUSD | ||
| #else | ||
| for (int i = 0; i < nb; i++) { | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Pre-existing, but this PR rewrites the exact #if/#else around it: this scalar branch is byte-for-byte identical to |
||
| const float d0 = GGML_CPU_FP16_TO_FP32(x[i].d); | ||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
This comment is not quite right: the block also uses XMM ops (the idx vectors and the movq load below), and VNNI plus 256-bit registers is not a sufficient condition, which is exactly how the gate above ended up too wide. Suggest: "uses only SSE/AVX2 ops (no 512-bit or AVX512-only instructions); requires AVX2". I would also drop the CPU model list, it is already stale (Meteor Lake, Sierra Forest).