strat9_kernel/framebuffer/x86/
sse2.rs1use core::arch::x86_64::*;
2
3#[inline]
4unsafe fn load_u32_unaligned(src: *const u8) -> i32 {
5 core::ptr::read_unaligned(src as *const u32) as i32
6}
7
8#[target_feature(enable = "sse2")]
9pub unsafe fn fill_sse2(dst: *mut u32, color: u32, count: usize) {
10 let color_vec = _mm_set1_epi32(color as i32);
14 let mut i = 0;
15
16 while i + 4 <= count {
17 _mm_storeu_si128(dst.add(i) as *mut __m128i, color_vec);
18 i += 4;
19 }
20
21 while i < count {
22 *dst.add(i) = color;
23 i += 1;
24 }
25}
26
27#[target_feature(enable = "sse2")]
28pub unsafe fn blit_sse2(dst: *mut u32, src: *const u32, count: usize) {
29 let mut i = 0;
30
31 while i + 4 <= count {
32 let src_vec = _mm_loadu_si128(src.add(i) as *const __m128i);
33 _mm_storeu_si128(dst.add(i) as *mut __m128i, src_vec);
34 i += 4;
35 }
36
37 while i < count {
38 *dst.add(i) = *src.add(i);
39 i += 1;
40 }
41}
42
43#[target_feature(enable = "sse2")]
46pub unsafe fn blend_sse2(dst: *mut u32, src: *const u32, alpha: u8, count: usize) {
47 let alpha_u16 = alpha as u16;
49 let inv_alpha = 255 - alpha_u16;
50
51 let alpha_vec = _mm_set1_epi16(alpha_u16 as i16);
52 let inv_alpha_vec = _mm_set1_epi16(inv_alpha as i16);
53 let zero = _mm_setzero_si128();
54 let ones = _mm_set1_epi16(1);
55
56 let mut i = 0;
57 while i + 4 <= count {
58 let d = _mm_loadu_si128(dst.add(i) as *const __m128i);
59 let s = _mm_loadu_si128(src.add(i) as *const __m128i);
60
61 let d_lo = _mm_unpacklo_epi8(d, zero);
62 let d_hi = _mm_unpackhi_epi8(d, zero);
63 let s_lo = _mm_unpacklo_epi8(s, zero);
64 let s_hi = _mm_unpackhi_epi8(s, zero);
65
66 let res_lo_s = _mm_mullo_epi16(s_lo, alpha_vec);
67 let res_lo_d = _mm_mullo_epi16(d_lo, inv_alpha_vec);
68 let res_lo = _mm_add_epi16(res_lo_s, res_lo_d);
69 let lo_div = _mm_srli_epi16(res_lo, 8);
71 let lo_corr = _mm_add_epi16(res_lo, _mm_add_epi16(lo_div, ones));
72 let res_lo_final = _mm_srli_epi16(lo_corr, 8);
73
74 let res_hi_s = _mm_mullo_epi16(s_hi, alpha_vec);
75 let res_hi_d = _mm_mullo_epi16(d_hi, inv_alpha_vec);
76 let res_hi = _mm_add_epi16(res_hi_s, res_hi_d);
77 let hi_div = _mm_srli_epi16(res_hi, 8);
78 let hi_corr = _mm_add_epi16(res_hi, _mm_add_epi16(hi_div, ones));
79 let res_hi_final = _mm_srli_epi16(hi_corr, 8);
80
81 let res = _mm_packus_epi16(res_lo_final, res_hi_final);
82 _mm_storeu_si128(dst.add(i) as *mut __m128i, res);
83 i += 4;
84 }
85
86 if i < count {
88 crate::framebuffer::generic::blend_generic(dst.add(i), src.add(i), alpha, count - i);
89 }
90}
91
92#[target_feature(enable = "sse2")]
93pub unsafe fn convert_bgr_to_argb_sse2(dst: *mut u32, src: *const u8, count: usize) {
94 crate::framebuffer::generic::convert_bgr_to_argb_generic(dst, src, count);
96}
97
98#[target_feature(enable = "ssse3")]
100unsafe fn convert_bgr_to_argb_ssse3(dst: *mut u32, src: *const u8, count: usize) {
101 #[rustfmt::skip]
106 let shuffle_mask = _mm_set_epi8(
107 -1, 11, 10, 9, -1, 8, 7, 6, -1, 5, 4, 3, -1, 2, 1, 0, );
112 let alpha_mask = _mm_set1_epi32(0xFF000000_u32 as i32);
114
115 let mut i = 0;
116 while i + 4 <= count {
118 let lo8 = _mm_loadl_epi64(src.add(i * 3) as *const __m128i);
120 let hi4 = _mm_cvtsi32_si128(load_u32_unaligned(src.add(i * 3 + 8)));
121 let raw = _mm_unpacklo_epi64(lo8, hi4);
122 let shuffled = _mm_shuffle_epi8(raw, shuffle_mask);
123 let result = _mm_or_si128(shuffled, alpha_mask);
124 _mm_storeu_si128(dst.add(i) as *mut __m128i, result);
125 i += 4;
126 }
127
128 if i < count {
130 crate::framebuffer::generic::convert_bgr_to_argb_generic(
131 dst.add(i),
132 src.add(i * 3),
133 count - i,
134 );
135 }
136}