Skip to main content

strat9_kernel/framebuffer/x86/
sse2.rs

1use core::arch::x86_64::*;
2
3#[inline]
4unsafe fn load_u32_unaligned(src: *const u8) -> i32 {
5    core::ptr::read_unaligned(src as *const u32) as i32
6}
7
8#[target_feature(enable = "sse2")]
9pub unsafe fn fill_sse2(dst: *mut u32, color: u32, count: usize) {
10    // Use the cached SSE2 implementation for every size until the NT
11    // toolchain path is restored. Never dispatch back into this function.
12
13    let color_vec = _mm_set1_epi32(color as i32);
14    let mut i = 0;
15
16    while i + 4 <= count {
17        _mm_storeu_si128(dst.add(i) as *mut __m128i, color_vec);
18        i += 4;
19    }
20
21    while i < count {
22        *dst.add(i) = color;
23        i += 1;
24    }
25}
26
27#[target_feature(enable = "sse2")]
28pub unsafe fn blit_sse2(dst: *mut u32, src: *const u32, count: usize) {
29    let mut i = 0;
30
31    while i + 4 <= count {
32        let src_vec = _mm_loadu_si128(src.add(i) as *const __m128i);
33        _mm_storeu_si128(dst.add(i) as *mut __m128i, src_vec);
34        i += 4;
35    }
36
37    while i < count {
38        *dst.add(i) = *src.add(i);
39        i += 1;
40    }
41}
42
43// In sse2, we use SSE4.1 blend where possible, but if only sse2 is available,
44// we'll implement a basic one. For simplicity, we use SSE2 compatible intrinsics.
45#[target_feature(enable = "sse2")]
46pub unsafe fn blend_sse2(dst: *mut u32, src: *const u32, alpha: u8, count: usize) {
47    // Basic SSE2 implementation for blend
48    let alpha_u16 = alpha as u16;
49    let inv_alpha = 255 - alpha_u16;
50
51    let alpha_vec = _mm_set1_epi16(alpha_u16 as i16);
52    let inv_alpha_vec = _mm_set1_epi16(inv_alpha as i16);
53    let zero = _mm_setzero_si128();
54    let ones = _mm_set1_epi16(1);
55
56    let mut i = 0;
57    while i + 4 <= count {
58        let d = _mm_loadu_si128(dst.add(i) as *const __m128i);
59        let s = _mm_loadu_si128(src.add(i) as *const __m128i);
60
61        let d_lo = _mm_unpacklo_epi8(d, zero);
62        let d_hi = _mm_unpackhi_epi8(d, zero);
63        let s_lo = _mm_unpacklo_epi8(s, zero);
64        let s_hi = _mm_unpackhi_epi8(s, zero);
65
66        let res_lo_s = _mm_mullo_epi16(s_lo, alpha_vec);
67        let res_lo_d = _mm_mullo_epi16(d_lo, inv_alpha_vec);
68        let res_lo = _mm_add_epi16(res_lo_s, res_lo_d);
69        // Exact /255: (x + (x>>8) + 1) >> 8  (dav1d formula)
70        let lo_div = _mm_srli_epi16(res_lo, 8);
71        let lo_corr = _mm_add_epi16(res_lo, _mm_add_epi16(lo_div, ones));
72        let res_lo_final = _mm_srli_epi16(lo_corr, 8);
73
74        let res_hi_s = _mm_mullo_epi16(s_hi, alpha_vec);
75        let res_hi_d = _mm_mullo_epi16(d_hi, inv_alpha_vec);
76        let res_hi = _mm_add_epi16(res_hi_s, res_hi_d);
77        let hi_div = _mm_srli_epi16(res_hi, 8);
78        let hi_corr = _mm_add_epi16(res_hi, _mm_add_epi16(hi_div, ones));
79        let res_hi_final = _mm_srli_epi16(hi_corr, 8);
80
81        let res = _mm_packus_epi16(res_lo_final, res_hi_final);
82        _mm_storeu_si128(dst.add(i) as *mut __m128i, res);
83        i += 4;
84    }
85
86    // fallback
87    if i < count {
88        crate::framebuffer::generic::blend_generic(dst.add(i), src.add(i), alpha, count - i);
89    }
90}
91
92#[target_feature(enable = "sse2")]
93pub unsafe fn convert_bgr_to_argb_sse2(dst: *mut u32, src: *const u8, count: usize) {
94    // TOOLCHAIN BUG (see x86/mod.rs): 128-bit pshufb aborts ISel here too.
95    crate::framebuffer::generic::convert_bgr_to_argb_generic(dst, src, count);
96}
97
98/// BGR24 => ARGB32 conversion using SSSE3 pshufb (4 pixels/iter)
99#[target_feature(enable = "ssse3")]
100unsafe fn convert_bgr_to_argb_ssse3(dst: *mut u32, src: *const u8, count: usize) {
101    // Shuffle mask: rearranges BGR BGR BGR BGR => B G R 0 B G R 0 B G R 0 B G R 0
102    // Input bytes:  0  1  2  3  4  5  6  7  8  9 10 11 12 13 14 15
103    //             [B0 G0 R0 B1 G1 R1 B2 G2 R2 B3 G3 R3  ?  ?  ?  ?]
104    // Output:      [B0 G0 R0  0 B1 G1 R1  0 B2 G2 R2  0 B3 G3 R3  0]
105    #[rustfmt::skip]
106    let shuffle_mask = _mm_set_epi8(
107        -1, 11, 10,  9,     // pixel 3: 0, R3, G3, B3
108        -1,  8,  7,  6,     // pixel 2: 0, R2, G2, B2
109        -1,  5,  4,  3,     // pixel 1: 0, R1, G1, B1
110        -1,  2,  1,  0,     // pixel 0: 0, R0, G0, B0
111    );
112    // Alpha mask: sets byte 3,7,11,15 to 0xFF
113    let alpha_mask = _mm_set1_epi32(0xFF000000_u32 as i32);
114
115    let mut i = 0;
116    // Charge exactement 12 octets (4 pixels BGR) sans dépassement de buffer
117    while i + 4 <= count {
118        // lecture sécurisée: 8 premiers octets (B0..R1) + 4 suivants (B2..R3)
119        let lo8 = _mm_loadl_epi64(src.add(i * 3) as *const __m128i);
120        let hi4 = _mm_cvtsi32_si128(load_u32_unaligned(src.add(i * 3 + 8)));
121        let raw = _mm_unpacklo_epi64(lo8, hi4);
122        let shuffled = _mm_shuffle_epi8(raw, shuffle_mask);
123        let result = _mm_or_si128(shuffled, alpha_mask);
124        _mm_storeu_si128(dst.add(i) as *mut __m128i, result);
125        i += 4;
126    }
127
128    // tail
129    if i < count {
130        crate::framebuffer::generic::convert_bgr_to_argb_generic(
131            dst.add(i),
132            src.add(i * 3),
133            count - i,
134        );
135    }
136}