FreeRDP
Loading...
Searching...
No Matches
prim_YUV_sse4.1.c
1
23#include <winpr/wtypes.h>
24#include <freerdp/config.h>
25
26#include <winpr/sysinfo.h>
27#include <winpr/crt.h>
28#include <freerdp/types.h>
29#include <freerdp/primitives.h>
30
31#include "prim_internal.h"
32#include "prim_avxsse.h"
33#include "prim_YUV.h"
34
35#if defined(SSE_AVX_INTRINSICS_ENABLED)
36#include <emmintrin.h>
37#include <tmmintrin.h>
38#include <smmintrin.h>
39
40static primitives_t* generic = nullptr;
41
42/****************************************************************************/
43/* sse41 YUV420 -> RGB conversion */
44/****************************************************************************/
45static inline __m128i* sse41_YUV444Pixel(__m128i* WINPR_RESTRICT dst, __m128i Yraw, __m128i Uraw,
46 __m128i Vraw, UINT8 pos)
47{
48 const __m128i mapY[] = { mm_set_epu32(0x80800380, 0x80800280, 0x80800180, 0x80800080),
49 mm_set_epu32(0x80800780, 0x80800680, 0x80800580, 0x80800480),
50 mm_set_epu32(0x80800B80, 0x80800A80, 0x80800980, 0x80800880),
51 mm_set_epu32(0x80800F80, 0x80800E80, 0x80800D80, 0x80800C80) };
52 const __m128i mapUV[] = { mm_set_epu32(0x80038002, 0x80018000, 0x80808080, 0x80808080),
53 mm_set_epu32(0x80078006, 0x80058004, 0x80808080, 0x80808080),
54 mm_set_epu32(0x800B800A, 0x80098008, 0x80808080, 0x80808080),
55 mm_set_epu32(0x800F800E, 0x800D800C, 0x80808080, 0x80808080) };
56 const __m128i mask[] = { mm_set_epu32(0x80038080, 0x80028080, 0x80018080, 0x80008080),
57 mm_set_epu32(0x80800380, 0x80800280, 0x80800180, 0x80800080),
58 mm_set_epu32(0x80808003, 0x80808002, 0x80808001, 0x80808000) };
59 const __m128i c128 = _mm_set1_epi16(128);
60 __m128i BGRX = _mm_and_si128(LOAD_SI128(dst),
61 mm_set_epu32(0xFF000000, 0xFF000000, 0xFF000000, 0xFF000000));
62 {
63 __m128i C;
64 __m128i D;
65 __m128i E;
66 /* Load Y values and expand to 32 bit */
67 {
68 C = _mm_shuffle_epi8(Yraw, mapY[pos]); /* Reorder and multiply by 256 */
69 }
70 /* Load U values and expand to 32 bit */
71 {
72 const __m128i U = _mm_shuffle_epi8(Uraw, mapUV[pos]); /* Reorder dcba */
73 D = _mm_sub_epi16(U, c128); /* D = U - 128 */
74 }
75 /* Load V values and expand to 32 bit */
76 {
77 const __m128i V = _mm_shuffle_epi8(Vraw, mapUV[pos]); /* Reorder dcba */
78 E = _mm_sub_epi16(V, c128); /* E = V - 128 */
79 }
80 /* Get the R value */
81 {
82 const __m128i c403 = _mm_set1_epi16(403);
83 const __m128i e403 =
84 _mm_unpackhi_epi16(_mm_mullo_epi16(E, c403), _mm_mulhi_epi16(E, c403));
85 const __m128i Rs = _mm_add_epi32(C, e403);
86 const __m128i R32 = _mm_srai_epi32(Rs, 8);
87 const __m128i R16 = _mm_packs_epi32(R32, _mm_setzero_si128());
88 const __m128i R = _mm_packus_epi16(R16, _mm_setzero_si128());
89 const __m128i packed = _mm_shuffle_epi8(R, mask[0]);
90 BGRX = _mm_or_si128(BGRX, packed);
91 }
92 /* Get the G value */
93 {
94 const __m128i c48 = _mm_set1_epi16(48);
95 const __m128i d48 =
96 _mm_unpackhi_epi16(_mm_mullo_epi16(D, c48), _mm_mulhi_epi16(D, c48));
97 const __m128i c120 = _mm_set1_epi16(120);
98 const __m128i e120 =
99 _mm_unpackhi_epi16(_mm_mullo_epi16(E, c120), _mm_mulhi_epi16(E, c120));
100 const __m128i de = _mm_add_epi32(d48, e120);
101 const __m128i Gs = _mm_sub_epi32(C, de);
102 const __m128i G32 = _mm_srai_epi32(Gs, 8);
103 const __m128i G16 = _mm_packs_epi32(G32, _mm_setzero_si128());
104 const __m128i G = _mm_packus_epi16(G16, _mm_setzero_si128());
105 const __m128i packed = _mm_shuffle_epi8(G, mask[1]);
106 BGRX = _mm_or_si128(BGRX, packed);
107 }
108 /* Get the B value */
109 {
110 const __m128i c475 = _mm_set1_epi16(475);
111 const __m128i d475 =
112 _mm_unpackhi_epi16(_mm_mullo_epi16(D, c475), _mm_mulhi_epi16(D, c475));
113 const __m128i Bs = _mm_add_epi32(C, d475);
114 const __m128i B32 = _mm_srai_epi32(Bs, 8);
115 const __m128i B16 = _mm_packs_epi32(B32, _mm_setzero_si128());
116 const __m128i B = _mm_packus_epi16(B16, _mm_setzero_si128());
117 const __m128i packed = _mm_shuffle_epi8(B, mask[2]);
118 BGRX = _mm_or_si128(BGRX, packed);
119 }
120 }
121 STORE_SI128(dst++, BGRX);
122 return dst;
123}
124
125static inline pstatus_t sse41_YUV420ToRGB_BGRX(const BYTE* WINPR_RESTRICT pSrc[],
126 const UINT32* WINPR_RESTRICT srcStep,
127 BYTE* WINPR_RESTRICT pDst, UINT32 dstStep,
128 const prim_size_t* WINPR_RESTRICT roi)
129{
130 const UINT32 nWidth = roi->width;
131 const UINT32 nHeight = roi->height;
132 const UINT32 pad = roi->width % 16;
133 const __m128i duplicate = _mm_set_epi8(7, 7, 6, 6, 5, 5, 4, 4, 3, 3, 2, 2, 1, 1, 0, 0);
134
135 for (size_t y = 0; y < nHeight; y++)
136 {
137 __m128i* dst = WINPR_PACKED_ALIGN_CAST(__m128i*, (pDst + dstStep * y));
138 const BYTE* YData = pSrc[0] + y * srcStep[0];
139 const BYTE* UData = pSrc[1] + (y / 2) * srcStep[1];
140 const BYTE* VData = pSrc[2] + (y / 2) * srcStep[2];
141
142 for (UINT32 x = 0; x < nWidth - pad; x += 16)
143 {
144 const __m128i Y = LOAD_SI128(YData);
145 const __m128i uRaw = LOAD_SI128(UData);
146 const __m128i vRaw = LOAD_SI128(VData);
147 const __m128i U = _mm_shuffle_epi8(uRaw, duplicate);
148 const __m128i V = _mm_shuffle_epi8(vRaw, duplicate);
149 YData += 16;
150 UData += 8;
151 VData += 8;
152 dst = sse41_YUV444Pixel(dst, Y, U, V, 0);
153 dst = sse41_YUV444Pixel(dst, Y, U, V, 1);
154 dst = sse41_YUV444Pixel(dst, Y, U, V, 2);
155 dst = sse41_YUV444Pixel(dst, Y, U, V, 3);
156 }
157
158 for (UINT32 x = 0; x < pad; x++)
159 {
160 const BYTE Y = *YData++;
161 const BYTE U = *UData;
162 const BYTE V = *VData;
163 dst = WINPR_PACKED_ALIGN_CAST(
164 __m128i*, writeYUVPixel((BYTE*)dst, PIXEL_FORMAT_BGRX32, Y, U, V, writePixelBGRX));
165
166 if (x % 2)
167 {
168 UData++;
169 VData++;
170 }
171 }
172 }
173
174 return PRIMITIVES_SUCCESS;
175}
176
177static pstatus_t sse41_YUV420ToRGB(const BYTE* WINPR_RESTRICT pSrc[3], const UINT32 srcStep[3],
178 BYTE* WINPR_RESTRICT pDst, UINT32 dstStep, UINT32 DstFormat,
179 const prim_size_t* WINPR_RESTRICT roi)
180{
181 switch (DstFormat)
182 {
183 case PIXEL_FORMAT_BGRX32:
184 case PIXEL_FORMAT_BGRA32:
185 return sse41_YUV420ToRGB_BGRX(pSrc, srcStep, pDst, dstStep, roi);
186
187 default:
188 return generic->YUV420ToRGB_8u_P3AC4R(pSrc, srcStep, pDst, dstStep, DstFormat, roi);
189 }
190}
191
192static inline void BGRX_fillRGB(size_t offset, BYTE* WINPR_RESTRICT pRGB[2],
193 const BYTE* WINPR_RESTRICT pY[2], const BYTE* WINPR_RESTRICT pU[2],
194 const BYTE* WINPR_RESTRICT pV[2], BOOL filter)
195{
196 WINPR_ASSERT(pRGB);
197 WINPR_ASSERT(pY);
198 WINPR_ASSERT(pU);
199 WINPR_ASSERT(pV);
200
201 const UINT32 DstFormat = PIXEL_FORMAT_BGRX32;
202 const UINT32 bpp = 4;
203
204 for (size_t i = 0; i < 2; i++)
205 {
206 for (size_t j = 0; j < 2; j++)
207 {
208 const BYTE Y = pY[i][offset + j];
209 BYTE U = pU[i][offset + j];
210 BYTE V = pV[i][offset + j];
211 if ((i == 0) && (j == 0) && filter)
212 {
213 const INT32 avgU =
214 4 * pU[0][offset] - pU[0][offset + 1] - pU[1][offset] - pU[1][offset + 1];
215 const INT32 avgV =
216 4 * pV[0][offset] - pV[0][offset + 1] - pV[1][offset] - pV[1][offset + 1];
217
218 U = CONDITIONAL_CLIP(avgU, pU[0][offset]);
219 V = CONDITIONAL_CLIP(avgV, pV[0][offset]);
220 }
221
222 writeYUVPixel(&pRGB[i][(j + offset) * bpp], DstFormat, Y, U, V, writePixelBGRX);
223 }
224 }
225}
226
227/* input are uint16_t vectors */
228static inline __m128i sse41_yuv2x_single(const __m128i Y, __m128i U, __m128i V, const short iMulU,
229 const short iMulV)
230{
231 const __m128i zero = _mm_set1_epi8(0);
232
233 __m128i Ylo = _mm_unpacklo_epi16(Y, zero);
234 __m128i Yhi = _mm_unpackhi_epi16(Y, zero);
235 if (iMulU != 0)
236 {
237 const __m128i addX = _mm_set1_epi16(128);
238 const __m128i D = _mm_sub_epi16(U, addX);
239 const __m128i mulU = _mm_set1_epi16(iMulU);
240 const __m128i mulDlo = _mm_mullo_epi16(D, mulU);
241 const __m128i mulDhi = _mm_mulhi_epi16(D, mulU);
242 const __m128i Dlo = _mm_unpacklo_epi16(mulDlo, mulDhi);
243 Ylo = _mm_add_epi32(Ylo, Dlo);
244
245 const __m128i Dhi = _mm_unpackhi_epi16(mulDlo, mulDhi);
246 Yhi = _mm_add_epi32(Yhi, Dhi);
247 }
248 if (iMulV != 0)
249 {
250 const __m128i addX = _mm_set1_epi16(128);
251 const __m128i E = _mm_sub_epi16(V, addX);
252 const __m128i mul = _mm_set1_epi16(iMulV);
253 const __m128i mulElo = _mm_mullo_epi16(E, mul);
254 const __m128i mulEhi = _mm_mulhi_epi16(E, mul);
255 const __m128i Elo = _mm_unpacklo_epi16(mulElo, mulEhi);
256 const __m128i esumlo = _mm_add_epi32(Ylo, Elo);
257
258 const __m128i Ehi = _mm_unpackhi_epi16(mulElo, mulEhi);
259 const __m128i esumhi = _mm_add_epi32(Yhi, Ehi);
260 Ylo = esumlo;
261 Yhi = esumhi;
262 }
263
264 const __m128i rYlo = _mm_srai_epi32(Ylo, 8);
265 const __m128i rYhi = _mm_srai_epi32(Yhi, 8);
266 const __m128i rY = _mm_packs_epi32(rYlo, rYhi);
267 return rY;
268}
269
270/* Input are uint8_t vectors */
271static inline __m128i sse41_yuv2x(const __m128i Y, __m128i U, __m128i V, const short iMulU,
272 const short iMulV)
273{
274 const __m128i zero = _mm_set1_epi8(0);
275
276 /* Ylo = Y * 256
277 * Ulo = uint8_t -> uint16_t
278 * Vlo = uint8_t -> uint16_t
279 */
280 const __m128i Ylo = _mm_unpacklo_epi8(zero, Y);
281 const __m128i Ulo = _mm_unpacklo_epi8(U, zero);
282 const __m128i Vlo = _mm_unpacklo_epi8(V, zero);
283 const __m128i preslo = sse41_yuv2x_single(Ylo, Ulo, Vlo, iMulU, iMulV);
284
285 const __m128i Yhi = _mm_unpackhi_epi8(zero, Y);
286 const __m128i Uhi = _mm_unpackhi_epi8(U, zero);
287 const __m128i Vhi = _mm_unpackhi_epi8(V, zero);
288 const __m128i preshi = sse41_yuv2x_single(Yhi, Uhi, Vhi, iMulU, iMulV);
289 const __m128i res = _mm_packus_epi16(preslo, preshi);
290
291 return res;
292}
293
294/* const INT32 r = ((256L * C(Y) + 0L * D(U) + 403L * E(V))) >> 8; */
295static inline __m128i sse41_yuv2r(const __m128i Y, __m128i U, __m128i V)
296{
297 return sse41_yuv2x(Y, U, V, 0, 403);
298}
299
300/* const INT32 g = ((256L * C(Y) - 48L * D(U) - 120L * E(V))) >> 8; */
301static inline __m128i sse41_yuv2g(const __m128i Y, __m128i U, __m128i V)
302{
303 return sse41_yuv2x(Y, U, V, -48, -120);
304}
305
306/* const INT32 b = ((256L * C(Y) + 475L * D(U) + 0L * E(V))) >> 8; */
307static inline __m128i sse41_yuv2b(const __m128i Y, __m128i U, __m128i V)
308{
309 return sse41_yuv2x(Y, U, V, 475, 0);
310}
311
312static inline void sse41_BGRX_fillRGB_pixel(BYTE* WINPR_RESTRICT pRGB, __m128i Y, __m128i U,
313 __m128i V)
314{
315 const __m128i zero = _mm_set1_epi8(0);
316 /* Y * 256 */
317 const __m128i r = sse41_yuv2r(Y, U, V);
318 const __m128i rx[2] = { _mm_unpackhi_epi8(r, zero), _mm_unpacklo_epi8(r, zero) };
319
320 const __m128i g = sse41_yuv2g(Y, U, V);
321 const __m128i b = sse41_yuv2b(Y, U, V);
322
323 const __m128i bg[2] = { _mm_unpackhi_epi8(b, g), _mm_unpacklo_epi8(b, g) };
324
325 const __m128i mask = mm_set_epu8(0x00, 0xFF, 0xFF, 0xFF, 0x00, 0xFF, 0xFF, 0xFF, 0x00, 0xFF,
326 0xFF, 0xFF, 0x00, 0xFF, 0xFF, 0xFF);
327
328 __m128i* rgb = WINPR_PACKED_ALIGN_CAST(__m128i*, pRGB);
329 const __m128i bgrx0 = _mm_unpacklo_epi16(bg[1], rx[1]);
330 _mm_maskmoveu_si128(bgrx0, mask, (char*)&rgb[0]);
331 const __m128i bgrx1 = _mm_unpackhi_epi16(bg[1], rx[1]);
332 _mm_maskmoveu_si128(bgrx1, mask, (char*)&rgb[1]);
333 const __m128i bgrx2 = _mm_unpacklo_epi16(bg[0], rx[0]);
334 _mm_maskmoveu_si128(bgrx2, mask, (char*)&rgb[2]);
335 const __m128i bgrx3 = _mm_unpackhi_epi16(bg[0], rx[0]);
336 _mm_maskmoveu_si128(bgrx3, mask, (char*)&rgb[3]);
337}
338
339static inline __m128i odd1sum(__m128i u1)
340{
341 const __m128i zero = _mm_set1_epi8(0);
342 const __m128i u1hi = _mm_unpackhi_epi8(u1, zero);
343 const __m128i u1lo = _mm_unpacklo_epi8(u1, zero);
344 return _mm_hadds_epi16(u1lo, u1hi);
345}
346
347static inline __m128i odd0sum(__m128i u0, __m128i u1sum)
348{
349 /* Mask out even bytes, extend uint8_t to uint16_t by filling in zero bytes,
350 * horizontally add the values */
351 const __m128i mask = mm_set_epu8(0x80, 0x0F, 0x80, 0x0D, 0x80, 0x0B, 0x80, 0x09, 0x80, 0x07,
352 0x80, 0x05, 0x80, 0x03, 0x80, 0x01);
353 const __m128i u0odd = _mm_shuffle_epi8(u0, mask);
354 return _mm_adds_epi16(u1sum, u0odd);
355}
356
357static inline __m128i calcavg(__m128i u0even, __m128i sum)
358{
359 const __m128i u4zero = _mm_slli_epi16(u0even, 2);
360 const __m128i uavg = _mm_sub_epi16(u4zero, sum);
361 const __m128i zero = _mm_set1_epi8(0);
362 const __m128i savg = _mm_packus_epi16(uavg, zero);
363 const __m128i smask = mm_set_epu8(0x80, 0x07, 0x80, 0x06, 0x80, 0x05, 0x80, 0x04, 0x80, 0x03,
364 0x80, 0x02, 0x80, 0x01, 0x80, 0x00);
365 return _mm_shuffle_epi8(savg, smask);
366}
367
368static inline __m128i diffmask(__m128i avg, __m128i u0even)
369{
370 /* Check for values >= 30 to apply the avg value to
371 * use int16 for calculations to avoid issues with signed 8bit integers
372 */
373 const __m128i diff = _mm_subs_epi16(u0even, avg);
374 const __m128i absdiff = _mm_abs_epi16(diff);
375 const __m128i val30 = _mm_set1_epi16(30);
376 return _mm_cmplt_epi16(absdiff, val30);
377}
378
379static inline void sse41_filter(__m128i pU[2])
380{
381 const __m128i u1sum = odd1sum(pU[1]);
382 const __m128i sum = odd0sum(pU[0], u1sum);
383
384 /* Mask out the odd bytes. We don“t need to do anything to make the uint8_t to uint16_t */
385 const __m128i emask = mm_set_epu8(0x00, 0xff, 0x00, 0xff, 0x00, 0xff, 0x00, 0xff, 0x00, 0xff,
386 0x00, 0xff, 0x00, 0xff, 0x00, 0xff);
387 const __m128i u0even = _mm_and_si128(pU[0], emask);
388 const __m128i avg = calcavg(u0even, sum);
389 const __m128i umask = diffmask(avg, u0even);
390
391 const __m128i u0orig = _mm_and_si128(u0even, umask);
392 const __m128i u0avg = _mm_andnot_si128(umask, avg);
393 const __m128i evenresult = _mm_or_si128(u0orig, u0avg);
394 const __m128i omask = mm_set_epu8(0xFF, 0x00, 0xFF, 0x00, 0xFF, 0x00, 0xFF, 0x00, 0xFF, 0x00,
395 0xFF, 0x00, 0xFF, 0x00, 0xFF, 0x00);
396 const __m128i u0odd = _mm_and_si128(pU[0], omask);
397 const __m128i result = _mm_or_si128(evenresult, u0odd);
398 pU[0] = result;
399}
400
401static inline void sse41_BGRX_fillRGB(BYTE* WINPR_RESTRICT pRGB[2], const __m128i pY[2],
402 __m128i pU[2], __m128i pV[2])
403{
404 WINPR_ASSERT(pRGB);
405 WINPR_ASSERT(pY);
406 WINPR_ASSERT(pU);
407 WINPR_ASSERT(pV);
408
409 sse41_filter(pU);
410 sse41_filter(pV);
411
412 for (size_t i = 0; i < 2; i++)
413 {
414 sse41_BGRX_fillRGB_pixel(pRGB[i], pY[i], pU[i], pV[i]);
415 }
416}
417
418static inline pstatus_t sse41_YUV444ToRGB_8u_P3AC4R_BGRX_DOUBLE_ROW(
419 BYTE* WINPR_RESTRICT pDst[2], const BYTE* WINPR_RESTRICT YData[2],
420 const BYTE* WINPR_RESTRICT UData[2], const BYTE* WINPR_RESTRICT VData[2], UINT32 nWidth)
421{
422 const UINT32 pad = nWidth % 16;
423
424 size_t x = 0;
425 for (; x < nWidth - pad; x += 16)
426 {
427 const __m128i Y[] = { LOAD_SI128(&YData[0][x]), LOAD_SI128(&YData[1][x]) };
428 __m128i U[] = { LOAD_SI128(&UData[0][x]), LOAD_SI128(&UData[1][x]) };
429 __m128i V[] = { LOAD_SI128(&VData[0][x]), LOAD_SI128(&VData[1][x]) };
430
431 BYTE* dstp[] = { &pDst[0][x * 4], &pDst[1][x * 4] };
432 sse41_BGRX_fillRGB(dstp, Y, U, V);
433 }
434
435 for (; x < nWidth; x += 2)
436 {
437 BGRX_fillRGB(x, pDst, YData, UData, VData, TRUE);
438 }
439
440 return PRIMITIVES_SUCCESS;
441}
442
443static inline void BGRX_fillRGB_single(size_t offset, BYTE* WINPR_RESTRICT pRGB,
444 const BYTE* WINPR_RESTRICT pY, const BYTE* WINPR_RESTRICT pU,
445 const BYTE* WINPR_RESTRICT pV, WINPR_ATTR_UNUSED BOOL filter)
446{
447 WINPR_ASSERT(pRGB);
448 WINPR_ASSERT(pY);
449 WINPR_ASSERT(pU);
450 WINPR_ASSERT(pV);
451
452 const UINT32 bpp = 4;
453
454 for (size_t j = 0; j < 2; j++)
455 {
456 const BYTE Y = pY[offset + j];
457 BYTE U = pU[offset + j];
458 BYTE V = pV[offset + j];
459
460 writeYUVPixel(&pRGB[(j + offset) * bpp], PIXEL_FORMAT_BGRX32, Y, U, V, writePixelBGRX);
461 }
462}
463
464static inline pstatus_t sse41_YUV444ToRGB_8u_P3AC4R_BGRX_SINGLE_ROW(
465 BYTE* WINPR_RESTRICT pDst, const BYTE* WINPR_RESTRICT YData, const BYTE* WINPR_RESTRICT UData,
466 const BYTE* WINPR_RESTRICT VData, UINT32 nWidth)
467{
468 for (size_t x = 0; x < nWidth; x += 2)
469 {
470 BGRX_fillRGB_single(x, pDst, YData, UData, VData, TRUE);
471 }
472
473 return PRIMITIVES_SUCCESS;
474}
475
476static inline pstatus_t sse41_YUV444ToRGB_8u_P3AC4R_BGRX(const BYTE* WINPR_RESTRICT pSrc[],
477 const UINT32 srcStep[],
478 BYTE* WINPR_RESTRICT pDst, UINT32 dstStep,
479 const prim_size_t* WINPR_RESTRICT roi)
480{
481 const UINT32 nWidth = roi->width;
482 const UINT32 nHeight = roi->height;
483
484 size_t y = 0;
485 for (; y < nHeight - nHeight % 2; y += 2)
486 {
487 BYTE* dst[] = { (pDst + dstStep * y), (pDst + dstStep * (y + 1)) };
488 const BYTE* YData[] = { pSrc[0] + y * srcStep[0], pSrc[0] + (y + 1) * srcStep[0] };
489 const BYTE* UData[] = { pSrc[1] + y * srcStep[1], pSrc[1] + (y + 1) * srcStep[1] };
490 const BYTE* VData[] = { pSrc[2] + y * srcStep[2], pSrc[2] + (y + 1) * srcStep[2] };
491
492 const pstatus_t rc =
493 sse41_YUV444ToRGB_8u_P3AC4R_BGRX_DOUBLE_ROW(dst, YData, UData, VData, nWidth);
494 if (rc != PRIMITIVES_SUCCESS)
495 return rc;
496 }
497 for (; y < nHeight; y++)
498 {
499 BYTE* dst = (pDst + dstStep * y);
500 const BYTE* YData = pSrc[0] + y * srcStep[0];
501 const BYTE* UData = pSrc[1] + y * srcStep[1];
502 const BYTE* VData = pSrc[2] + y * srcStep[2];
503 const pstatus_t rc =
504 sse41_YUV444ToRGB_8u_P3AC4R_BGRX_SINGLE_ROW(dst, YData, UData, VData, nWidth);
505 if (rc != PRIMITIVES_SUCCESS)
506 return rc;
507 }
508
509 return PRIMITIVES_SUCCESS;
510}
511
512static pstatus_t sse41_YUV444ToRGB_8u_P3AC4R(const BYTE* WINPR_RESTRICT pSrc[],
513 const UINT32 srcStep[], BYTE* WINPR_RESTRICT pDst,
514 UINT32 dstStep, UINT32 DstFormat,
515 const prim_size_t* WINPR_RESTRICT roi)
516{
517 switch (DstFormat)
518 {
519 case PIXEL_FORMAT_BGRX32:
520 case PIXEL_FORMAT_BGRA32:
521 return sse41_YUV444ToRGB_8u_P3AC4R_BGRX(pSrc, srcStep, pDst, dstStep, roi);
522
523 default:
524 return generic->YUV444ToRGB_8u_P3AC4R(pSrc, srcStep, pDst, dstStep, DstFormat, roi);
525 }
526}
527
528/****************************************************************************/
529/* sse41 RGB -> YUV420 conversion **/
530/****************************************************************************/
531
553#define BGRX_Y_FACTORS _mm_set_epi8(0, 27, 92, 9, 0, 27, 92, 9, 0, 27, 92, 9, 0, 27, 92, 9)
554#define BGRX_U_FACTORS \
555 _mm_set_epi8(0, -29, -99, 127, 0, -29, -99, 127, 0, -29, -99, 127, 0, -29, -99, 127)
556#define BGRX_V_FACTORS \
557 _mm_set_epi8(0, 127, -116, -12, 0, 127, -116, -12, 0, 127, -116, -12, 0, 127, -116, -12)
558#define CONST128_FACTORS _mm_set1_epi8(-128)
559
560#define Y_SHIFT 7
561#define U_SHIFT 8
562#define V_SHIFT 8
563
564/*
565TODO:
566RGB[AX] can simply be supported using the following factors. And instead of loading the
567globals directly the functions below could be passed pointers to the correct vectors
568depending on the source picture format.
569
570PRIM_ALIGN_128 static const BYTE rgbx_y_factors[] = {
571 27, 92, 9, 0, 27, 92, 9, 0, 27, 92, 9, 0, 27, 92, 9, 0
572};
573PRIM_ALIGN_128 static const BYTE rgbx_u_factors[] = {
574 -15, -49, 64, 0, -15, -49, 64, 0, -15, -49, 64, 0, -15, -49, 64, 0
575};
576PRIM_ALIGN_128 static const BYTE rgbx_v_factors[] = {
577 64, -58, -6, 0, 64, -58, -6, 0, 64, -58, -6, 0, 64, -58, -6, 0
578};
579*/
580
581static inline void sse41_BGRX_TO_YUV(const BYTE* WINPR_RESTRICT pLine1, BYTE* WINPR_RESTRICT pYLine,
582 BYTE* WINPR_RESTRICT pULine, BYTE* WINPR_RESTRICT pVLine)
583{
584 const BYTE r1 = pLine1[2];
585 const BYTE g1 = pLine1[1];
586 const BYTE b1 = pLine1[0];
587
588 if (pYLine)
589 pYLine[0] = RGB2Y(r1, g1, b1);
590 if (pULine)
591 pULine[0] = RGB2U(r1, g1, b1);
592 if (pVLine)
593 pVLine[0] = RGB2V(r1, g1, b1);
594}
595
596/* compute the luma (Y) component from a single rgb source line */
597
598static inline void sse41_RGBToYUV420_BGRX_Y(const BYTE* WINPR_RESTRICT src, BYTE* dst, UINT32 width)
599{
600 const __m128i y_factors = BGRX_Y_FACTORS;
601 const __m128i* argb = WINPR_PACKED_ALIGN_CAST(const __m128i*, src);
602 __m128i* ydst = WINPR_PACKED_ALIGN_CAST(__m128i*, dst);
603
604 UINT32 x = 0;
605
606 for (; x < width - width % 16; x += 16)
607 {
608 /* store 16 rgba pixels in 4 128 bit registers */
609 __m128i x0 = LOAD_SI128(argb++); // 1st 4 pixels
610 {
611 x0 = _mm_maddubs_epi16(x0, y_factors);
612
613 __m128i x1 = LOAD_SI128(argb++); // 2nd 4 pixels
614 x1 = _mm_maddubs_epi16(x1, y_factors);
615 x0 = _mm_hadds_epi16(x0, x1);
616 x0 = _mm_srli_epi16(x0, Y_SHIFT);
617 }
618
619 __m128i x2 = LOAD_SI128(argb++); // 3rd 4 pixels
620 {
621 x2 = _mm_maddubs_epi16(x2, y_factors);
622
623 __m128i x3 = LOAD_SI128(argb++); // 4th 4 pixels
624 x3 = _mm_maddubs_epi16(x3, y_factors);
625 x2 = _mm_hadds_epi16(x2, x3);
626 x2 = _mm_srli_epi16(x2, Y_SHIFT);
627 }
628
629 x0 = _mm_packus_epi16(x0, x2);
630 /* save to y plane */
631 STORE_SI128(ydst++, x0);
632 }
633
634 for (; x < width; x++)
635 {
636 sse41_BGRX_TO_YUV(&src[4ULL * x], &dst[x], nullptr, nullptr);
637 }
638}
639
640/* compute the chrominance (UV) components from two rgb source lines */
641
642static inline void sse41_RGBToYUV420_BGRX_UV(const BYTE* WINPR_RESTRICT src1,
643 const BYTE* WINPR_RESTRICT src2,
644 BYTE* WINPR_RESTRICT dst1, BYTE* WINPR_RESTRICT dst2,
645 UINT32 width)
646{
647 const __m128i u_factors = BGRX_U_FACTORS;
648 const __m128i v_factors = BGRX_V_FACTORS;
649 const __m128i vector128 = CONST128_FACTORS;
650
651 size_t x = 0;
652
653 for (; x < width - width % 16; x += 16)
654 {
655 const __m128i* rgb1 = WINPR_PACKED_ALIGN_CAST(const __m128i*, &src1[4ULL * x]);
656 const __m128i* rgb2 = WINPR_PACKED_ALIGN_CAST(const __m128i*, &src2[4ULL * x]);
657 __m64* udst = WINPR_PACKED_ALIGN_CAST(__m64*, &dst1[x / 2]);
658 __m64* vdst = WINPR_PACKED_ALIGN_CAST(__m64*, &dst2[x / 2]);
659
660 /* subsample 16x2 pixels into 16x1 pixels */
661 __m128i x0 = LOAD_SI128(&rgb1[0]);
662 __m128i x4 = LOAD_SI128(&rgb2[0]);
663 x0 = _mm_avg_epu8(x0, x4);
664
665 __m128i x1 = LOAD_SI128(&rgb1[1]);
666 x4 = LOAD_SI128(&rgb2[1]);
667 x1 = _mm_avg_epu8(x1, x4);
668
669 __m128i x2 = LOAD_SI128(&rgb1[2]);
670 x4 = LOAD_SI128(&rgb2[2]);
671 x2 = _mm_avg_epu8(x2, x4);
672
673 __m128i x3 = LOAD_SI128(&rgb1[3]);
674 x4 = LOAD_SI128(&rgb2[3]);
675 x3 = _mm_avg_epu8(x3, x4);
676
677 /* subsample these 16x1 pixels into 8x1 pixels */
683 x4 = _mm_castps_si128(_mm_shuffle_ps(_mm_castsi128_ps(x0), _mm_castsi128_ps(x1), 0x88));
684 x0 = _mm_castps_si128(_mm_shuffle_ps(_mm_castsi128_ps(x0), _mm_castsi128_ps(x1), 0xdd));
685 x0 = _mm_avg_epu8(x0, x4);
686 x4 = _mm_castps_si128(_mm_shuffle_ps(_mm_castsi128_ps(x2), _mm_castsi128_ps(x3), 0x88));
687 x1 = _mm_castps_si128(_mm_shuffle_ps(_mm_castsi128_ps(x2), _mm_castsi128_ps(x3), 0xdd));
688 x1 = _mm_avg_epu8(x1, x4);
689 /* multiplications and subtotals */
690 x2 = _mm_maddubs_epi16(x0, u_factors);
691 x3 = _mm_maddubs_epi16(x1, u_factors);
692 x4 = _mm_maddubs_epi16(x0, v_factors);
693 __m128i x5 = _mm_maddubs_epi16(x1, v_factors);
694 /* the total sums */
695 x0 = _mm_hadd_epi16(x2, x3);
696 x1 = _mm_hadd_epi16(x4, x5);
697 /* shift the results */
698 x0 = _mm_srai_epi16(x0, U_SHIFT);
699 x1 = _mm_srai_epi16(x1, V_SHIFT);
700 /* pack the 16 words into bytes */
701 x0 = _mm_packs_epi16(x0, x1);
702 /* add 128 */
703 x0 = _mm_sub_epi8(x0, vector128);
704 /* the lower 8 bytes go to the u plane */
705 _mm_storel_pi(udst, _mm_castsi128_ps(x0));
706 /* the upper 8 bytes go to the v plane */
707 _mm_storeh_pi(vdst, _mm_castsi128_ps(x0));
708 }
709
710 for (; x < width - width % 2; x += 2)
711 {
712 BYTE u[4] = WINPR_C_ARRAY_INIT;
713 BYTE v[4] = WINPR_C_ARRAY_INIT;
714 sse41_BGRX_TO_YUV(&src1[4ULL * x], nullptr, &u[0], &v[0]);
715 sse41_BGRX_TO_YUV(&src1[4ULL * (1ULL + x)], nullptr, &u[1], &v[1]);
716 sse41_BGRX_TO_YUV(&src2[4ULL * x], nullptr, &u[2], &v[2]);
717 sse41_BGRX_TO_YUV(&src2[4ULL * (1ULL + x)], nullptr, &u[3], &v[3]);
718 const INT16 u4 = WINPR_ASSERTING_INT_CAST(INT16, (INT16)u[0] + u[1] + u[2] + u[3]);
719 const INT16 uu = WINPR_ASSERTING_INT_CAST(INT16, u4 / 4);
720 const BYTE u8 = CLIP(uu);
721 dst1[x / 2] = u8;
722
723 const INT16 v4 = WINPR_ASSERTING_INT_CAST(INT16, (INT16)v[0] + v[1] + v[2] + v[3]);
724 const INT16 vu = WINPR_ASSERTING_INT_CAST(INT16, v4 / 4);
725 const BYTE v8 = CLIP(vu);
726 dst2[x / 2] = v8;
727 }
728}
729
730static pstatus_t sse41_RGBToYUV420_BGRX(const BYTE* WINPR_RESTRICT pSrc, UINT32 srcStep,
731 BYTE* WINPR_RESTRICT pDst[], const UINT32 dstStep[],
732 const prim_size_t* WINPR_RESTRICT roi)
733{
734 if (roi->height < 1 || roi->width < 1)
735 {
736 return !PRIMITIVES_SUCCESS;
737 }
738
739 size_t y = 0;
740 for (; y < roi->height - roi->height % 2; y += 2)
741 {
742 const BYTE* line1 = &pSrc[y * srcStep];
743 const BYTE* line2 = &pSrc[(1ULL + y) * srcStep];
744 BYTE* ydst1 = &pDst[0][y * dstStep[0]];
745 BYTE* ydst2 = &pDst[0][(1ULL + y) * dstStep[0]];
746 BYTE* udst = &pDst[1][y / 2 * dstStep[1]];
747 BYTE* vdst = &pDst[2][y / 2 * dstStep[2]];
748
749 sse41_RGBToYUV420_BGRX_UV(line1, line2, udst, vdst, roi->width);
750 sse41_RGBToYUV420_BGRX_Y(line1, ydst1, roi->width);
751 sse41_RGBToYUV420_BGRX_Y(line2, ydst2, roi->width);
752 }
753
754 for (; y < roi->height; y++)
755 {
756 const BYTE* line = &pSrc[y * srcStep];
757 BYTE* ydst = &pDst[0][1ULL * y * dstStep[0]];
758 sse41_RGBToYUV420_BGRX_Y(line, ydst, roi->width);
759 }
760
761 return PRIMITIVES_SUCCESS;
762}
763
764static pstatus_t sse41_RGBToYUV420(const BYTE* WINPR_RESTRICT pSrc, UINT32 srcFormat,
765 UINT32 srcStep, BYTE* WINPR_RESTRICT pDst[],
766 const UINT32 dstStep[], const prim_size_t* WINPR_RESTRICT roi)
767{
768 switch (srcFormat)
769 {
770 case PIXEL_FORMAT_BGRX32:
771 case PIXEL_FORMAT_BGRA32:
772 return sse41_RGBToYUV420_BGRX(pSrc, srcStep, pDst, dstStep, roi);
773
774 default:
775 return generic->RGBToYUV420_8u_P3AC4R(pSrc, srcFormat, srcStep, pDst, dstStep, roi);
776 }
777}
778
779/****************************************************************************/
780/* sse41 RGB -> AVC444-YUV conversion **/
781/****************************************************************************/
782
783static inline void sse41_RGBToAVC444YUV_BGRX_DOUBLE_ROW(
784 const BYTE* WINPR_RESTRICT srcEven, const BYTE* WINPR_RESTRICT srcOdd,
785 BYTE* WINPR_RESTRICT b1Even, BYTE* WINPR_RESTRICT b1Odd, BYTE* WINPR_RESTRICT b2,
786 BYTE* WINPR_RESTRICT b3, BYTE* WINPR_RESTRICT b4, BYTE* WINPR_RESTRICT b5,
787 BYTE* WINPR_RESTRICT b6, BYTE* WINPR_RESTRICT b7, UINT32 width)
788{
789 const __m128i* argbEven = WINPR_PACKED_ALIGN_CAST(const __m128i*, srcEven);
790 const __m128i* argbOdd = WINPR_PACKED_ALIGN_CAST(const __m128i*, srcOdd);
791 const __m128i y_factors = BGRX_Y_FACTORS;
792 const __m128i u_factors = BGRX_U_FACTORS;
793 const __m128i v_factors = BGRX_V_FACTORS;
794 const __m128i vector128 = CONST128_FACTORS;
795
796 UINT32 x = 0;
797 for (; x < width - width % 16; x += 16)
798 {
799 /* store 16 rgba pixels in 4 128 bit registers */
800 const __m128i xe1 = LOAD_SI128(argbEven++); // 1st 4 pixels
801 const __m128i xe2 = LOAD_SI128(argbEven++); // 2nd 4 pixels
802 const __m128i xe3 = LOAD_SI128(argbEven++); // 3rd 4 pixels
803 const __m128i xe4 = LOAD_SI128(argbEven++); // 4th 4 pixels
804 const __m128i xo1 = LOAD_SI128(argbOdd++); // 1st 4 pixels
805 const __m128i xo2 = LOAD_SI128(argbOdd++); // 2nd 4 pixels
806 const __m128i xo3 = LOAD_SI128(argbOdd++); // 3rd 4 pixels
807 const __m128i xo4 = LOAD_SI128(argbOdd++); // 4th 4 pixels
808 {
809 /* Y: multiplications with subtotals and horizontal sums */
810 const __m128i ye1 = _mm_srli_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe1, y_factors),
811 _mm_maddubs_epi16(xe2, y_factors)),
812 Y_SHIFT);
813 const __m128i ye2 = _mm_srli_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe3, y_factors),
814 _mm_maddubs_epi16(xe4, y_factors)),
815 Y_SHIFT);
816 const __m128i ye = _mm_packus_epi16(ye1, ye2);
817 const __m128i yo1 = _mm_srli_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo1, y_factors),
818 _mm_maddubs_epi16(xo2, y_factors)),
819 Y_SHIFT);
820 const __m128i yo2 = _mm_srli_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo3, y_factors),
821 _mm_maddubs_epi16(xo4, y_factors)),
822 Y_SHIFT);
823 const __m128i yo = _mm_packus_epi16(yo1, yo2);
824 /* store y [b1] */
825 STORE_SI128(b1Even, ye);
826 b1Even += 16;
827
828 if (b1Odd)
829 {
830 STORE_SI128(b1Odd, yo);
831 b1Odd += 16;
832 }
833 }
834 {
835 /* We have now
836 * 16 even U values in ue
837 * 16 odd U values in uo
838 *
839 * We need to split these according to
840 * 3.3.8.3.2 YUV420p Stream Combination for YUV444 mode */
841 __m128i ue;
842 __m128i uo = WINPR_C_ARRAY_INIT;
843 {
844 const __m128i ue1 =
845 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe1, u_factors),
846 _mm_maddubs_epi16(xe2, u_factors)),
847 U_SHIFT);
848 const __m128i ue2 =
849 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe3, u_factors),
850 _mm_maddubs_epi16(xe4, u_factors)),
851 U_SHIFT);
852 ue = _mm_sub_epi8(_mm_packs_epi16(ue1, ue2), vector128);
853 }
854
855 if (b1Odd)
856 {
857 const __m128i uo1 =
858 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo1, u_factors),
859 _mm_maddubs_epi16(xo2, u_factors)),
860 U_SHIFT);
861 const __m128i uo2 =
862 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo3, u_factors),
863 _mm_maddubs_epi16(xo4, u_factors)),
864 U_SHIFT);
865 uo = _mm_sub_epi8(_mm_packs_epi16(uo1, uo2), vector128);
866 }
867
868 /* Now we need the following storage distribution:
869 * 2x 2y -> b2
870 * x 2y+1 -> b4
871 * 2x+1 2y -> b6 */
872 if (b1Odd) /* b2 */
873 {
874 const __m128i ueh = _mm_unpackhi_epi8(ue, _mm_setzero_si128());
875 const __m128i uoh = _mm_unpackhi_epi8(uo, _mm_setzero_si128());
876 const __m128i hi = _mm_add_epi16(ueh, uoh);
877 const __m128i uel = _mm_unpacklo_epi8(ue, _mm_setzero_si128());
878 const __m128i uol = _mm_unpacklo_epi8(uo, _mm_setzero_si128());
879 const __m128i lo = _mm_add_epi16(uel, uol);
880 const __m128i added = _mm_hadd_epi16(lo, hi);
881 const __m128i avg16 = _mm_srai_epi16(added, 2);
882 const __m128i avg = _mm_packus_epi16(avg16, avg16);
883 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, b2), avg);
884 }
885 else
886 {
887 const __m128i mask =
888 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
889 (char)0x80, (char)0x80, (char)0x80, 14, 12, 10, 8, 6, 4, 2, 0);
890 const __m128i ud = _mm_shuffle_epi8(ue, mask);
891 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, b2), ud);
892 }
893
894 b2 += 8;
895
896 if (b1Odd) /* b4 */
897 {
898 STORE_SI128(b4, uo);
899 b4 += 16;
900 }
901
902 {
903 /* b6 */
904 const __m128i mask =
905 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
906 (char)0x80, (char)0x80, (char)0x80, 15, 13, 11, 9, 7, 5, 3, 1);
907 const __m128i ude = _mm_shuffle_epi8(ue, mask);
908 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, b6), ude);
909 b6 += 8;
910 }
911 }
912 {
913 /* We have now
914 * 16 even V values in ue
915 * 16 odd V values in uo
916 *
917 * We need to split these according to
918 * 3.3.8.3.2 YUV420p Stream Combination for YUV444 mode */
919 __m128i ve;
920 __m128i vo = WINPR_C_ARRAY_INIT;
921 {
922 const __m128i ve1 =
923 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe1, v_factors),
924 _mm_maddubs_epi16(xe2, v_factors)),
925 V_SHIFT);
926 const __m128i ve2 =
927 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe3, v_factors),
928 _mm_maddubs_epi16(xe4, v_factors)),
929 V_SHIFT);
930 ve = _mm_sub_epi8(_mm_packs_epi16(ve1, ve2), vector128);
931 }
932
933 if (b1Odd)
934 {
935 const __m128i vo1 =
936 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo1, v_factors),
937 _mm_maddubs_epi16(xo2, v_factors)),
938 V_SHIFT);
939 const __m128i vo2 =
940 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo3, v_factors),
941 _mm_maddubs_epi16(xo4, v_factors)),
942 V_SHIFT);
943 vo = _mm_sub_epi8(_mm_packs_epi16(vo1, vo2), vector128);
944 }
945
946 /* Now we need the following storage distribution:
947 * 2x 2y -> b3
948 * x 2y+1 -> b5
949 * 2x+1 2y -> b7 */
950 if (b1Odd) /* b3 */
951 {
952 const __m128i veh = _mm_unpackhi_epi8(ve, _mm_setzero_si128());
953 const __m128i voh = _mm_unpackhi_epi8(vo, _mm_setzero_si128());
954 const __m128i hi = _mm_add_epi16(veh, voh);
955 const __m128i vel = _mm_unpacklo_epi8(ve, _mm_setzero_si128());
956 const __m128i vol = _mm_unpacklo_epi8(vo, _mm_setzero_si128());
957 const __m128i lo = _mm_add_epi16(vel, vol);
958 const __m128i added = _mm_hadd_epi16(lo, hi);
959 const __m128i avg16 = _mm_srai_epi16(added, 2);
960 const __m128i avg = _mm_packus_epi16(avg16, avg16);
961 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, b3), avg);
962 }
963 else
964 {
965 const __m128i mask =
966 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
967 (char)0x80, (char)0x80, (char)0x80, 14, 12, 10, 8, 6, 4, 2, 0);
968 const __m128i vd = _mm_shuffle_epi8(ve, mask);
969 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, b3), vd);
970 }
971
972 b3 += 8;
973
974 if (b1Odd) /* b5 */
975 {
976 STORE_SI128(b5, vo);
977 b5 += 16;
978 }
979
980 {
981 /* b7 */
982 const __m128i mask =
983 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
984 (char)0x80, (char)0x80, (char)0x80, 15, 13, 11, 9, 7, 5, 3, 1);
985 const __m128i vde = _mm_shuffle_epi8(ve, mask);
986 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, b7), vde);
987 b7 += 8;
988 }
989 }
990 }
991
992 general_RGBToAVC444YUV_BGRX_DOUBLE_ROW(x, srcEven, srcOdd, b1Even, b1Odd, b2, b3, b4, b5, b6,
993 b7, width);
994}
995
996static pstatus_t sse41_RGBToAVC444YUV_BGRX(const BYTE* WINPR_RESTRICT pSrc,
997 WINPR_ATTR_UNUSED UINT32 srcFormat, UINT32 srcStep,
998 BYTE* WINPR_RESTRICT pDst1[], const UINT32 dst1Step[],
999 BYTE* WINPR_RESTRICT pDst2[], const UINT32 dst2Step[],
1000 const prim_size_t* WINPR_RESTRICT roi)
1001{
1002 if (roi->height < 1 || roi->width < 1)
1003 return !PRIMITIVES_SUCCESS;
1004
1005 size_t y = 0;
1006 for (; y < roi->height - roi->height % 2; y += 2)
1007 {
1008 const BYTE* srcEven = pSrc + y * srcStep;
1009 const BYTE* srcOdd = pSrc + (y + 1) * srcStep;
1010 const size_t i = y >> 1;
1011 const size_t n = (i & (size_t)~7) + i;
1012 BYTE* b1Even = pDst1[0] + y * dst1Step[0];
1013 BYTE* b1Odd = (b1Even + dst1Step[0]);
1014 BYTE* b2 = pDst1[1] + (y / 2) * dst1Step[1];
1015 BYTE* b3 = pDst1[2] + (y / 2) * dst1Step[2];
1016 BYTE* b4 = pDst2[0] + 1ULL * dst2Step[0] * n;
1017 BYTE* b5 = b4 + 8ULL * dst2Step[0];
1018 BYTE* b6 = pDst2[1] + (y / 2) * dst2Step[1];
1019 BYTE* b7 = pDst2[2] + (y / 2) * dst2Step[2];
1020 sse41_RGBToAVC444YUV_BGRX_DOUBLE_ROW(srcEven, srcOdd, b1Even, b1Odd, b2, b3, b4, b5, b6, b7,
1021 roi->width);
1022 }
1023
1024 for (; y < roi->height; y++)
1025 {
1026 const BYTE* srcEven = pSrc + y * srcStep;
1027 BYTE* b1Even = pDst1[0] + y * dst1Step[0];
1028 BYTE* b2 = pDst1[1] + (y / 2) * dst1Step[1];
1029 BYTE* b3 = pDst1[2] + (y / 2) * dst1Step[2];
1030 BYTE* b6 = pDst2[1] + (y / 2) * dst2Step[1];
1031 BYTE* b7 = pDst2[2] + (y / 2) * dst2Step[2];
1032 general_RGBToAVC444YUV_BGRX_DOUBLE_ROW(0, srcEven, nullptr, b1Even, nullptr, b2, b3,
1033 nullptr, nullptr, b6, b7, roi->width);
1034 }
1035
1036 return PRIMITIVES_SUCCESS;
1037}
1038
1039static pstatus_t sse41_RGBToAVC444YUV(const BYTE* WINPR_RESTRICT pSrc, UINT32 srcFormat,
1040 UINT32 srcStep, BYTE* WINPR_RESTRICT pDst1[],
1041 const UINT32 dst1Step[], BYTE* WINPR_RESTRICT pDst2[],
1042 const UINT32 dst2Step[],
1043 const prim_size_t* WINPR_RESTRICT roi)
1044{
1045 switch (srcFormat)
1046 {
1047 case PIXEL_FORMAT_BGRX32:
1048 case PIXEL_FORMAT_BGRA32:
1049 return sse41_RGBToAVC444YUV_BGRX(pSrc, srcFormat, srcStep, pDst1, dst1Step, pDst2,
1050 dst2Step, roi);
1051
1052 default:
1053 return generic->RGBToAVC444YUV(pSrc, srcFormat, srcStep, pDst1, dst1Step, pDst2,
1054 dst2Step, roi);
1055 }
1056}
1057
1058/* Mapping of arguments:
1059 *
1060 * b1 [even lines] -> yLumaDstEven
1061 * b1 [odd lines] -> yLumaDstOdd
1062 * b2 -> uLumaDst
1063 * b3 -> vLumaDst
1064 * b4 -> yChromaDst1
1065 * b5 -> yChromaDst2
1066 * b6 -> uChromaDst1
1067 * b7 -> uChromaDst2
1068 * b8 -> vChromaDst1
1069 * b9 -> vChromaDst2
1070 */
1071static inline void sse41_RGBToAVC444YUVv2_BGRX_DOUBLE_ROW(
1072 const BYTE* WINPR_RESTRICT srcEven, const BYTE* WINPR_RESTRICT srcOdd,
1073 BYTE* WINPR_RESTRICT yLumaDstEven, BYTE* WINPR_RESTRICT yLumaDstOdd,
1074 BYTE* WINPR_RESTRICT uLumaDst, BYTE* WINPR_RESTRICT vLumaDst,
1075 BYTE* WINPR_RESTRICT yEvenChromaDst1, BYTE* WINPR_RESTRICT yEvenChromaDst2,
1076 BYTE* WINPR_RESTRICT yOddChromaDst1, BYTE* WINPR_RESTRICT yOddChromaDst2,
1077 BYTE* WINPR_RESTRICT uChromaDst1, BYTE* WINPR_RESTRICT uChromaDst2,
1078 BYTE* WINPR_RESTRICT vChromaDst1, BYTE* WINPR_RESTRICT vChromaDst2, UINT32 width)
1079{
1080 const __m128i vector128 = CONST128_FACTORS;
1081 const __m128i* argbEven = WINPR_PACKED_ALIGN_CAST(const __m128i*, srcEven);
1082 const __m128i* argbOdd = WINPR_PACKED_ALIGN_CAST(const __m128i*, srcOdd);
1083
1084 UINT32 x = 0;
1085 for (; x < width - width % 16; x += 16)
1086 {
1087 /* store 16 rgba pixels in 4 128 bit registers
1088 * for even and odd rows.
1089 */
1090 const __m128i xe1 = LOAD_SI128(argbEven++); /* 1st 4 pixels */
1091 const __m128i xe2 = LOAD_SI128(argbEven++); /* 2nd 4 pixels */
1092 const __m128i xe3 = LOAD_SI128(argbEven++); /* 3rd 4 pixels */
1093 const __m128i xe4 = LOAD_SI128(argbEven++); /* 4th 4 pixels */
1094 const __m128i xo1 = LOAD_SI128(argbOdd++); /* 1st 4 pixels */
1095 const __m128i xo2 = LOAD_SI128(argbOdd++); /* 2nd 4 pixels */
1096 const __m128i xo3 = LOAD_SI128(argbOdd++); /* 3rd 4 pixels */
1097 const __m128i xo4 = LOAD_SI128(argbOdd++); /* 4th 4 pixels */
1098 {
1099 /* Y: multiplications with subtotals and horizontal sums */
1100 const __m128i y_factors = BGRX_Y_FACTORS;
1101 const __m128i ye1 = _mm_srli_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe1, y_factors),
1102 _mm_maddubs_epi16(xe2, y_factors)),
1103 Y_SHIFT);
1104 const __m128i ye2 = _mm_srli_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe3, y_factors),
1105 _mm_maddubs_epi16(xe4, y_factors)),
1106 Y_SHIFT);
1107 const __m128i ye = _mm_packus_epi16(ye1, ye2);
1108 /* store y [b1] */
1109 STORE_SI128(yLumaDstEven, ye);
1110 yLumaDstEven += 16;
1111 }
1112
1113 if (yLumaDstOdd)
1114 {
1115 const __m128i y_factors = BGRX_Y_FACTORS;
1116 const __m128i yo1 = _mm_srli_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo1, y_factors),
1117 _mm_maddubs_epi16(xo2, y_factors)),
1118 Y_SHIFT);
1119 const __m128i yo2 = _mm_srli_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo3, y_factors),
1120 _mm_maddubs_epi16(xo4, y_factors)),
1121 Y_SHIFT);
1122 const __m128i yo = _mm_packus_epi16(yo1, yo2);
1123 STORE_SI128(yLumaDstOdd, yo);
1124 yLumaDstOdd += 16;
1125 }
1126
1127 {
1128 /* We have now
1129 * 16 even U values in ue
1130 * 16 odd U values in uo
1131 *
1132 * We need to split these according to
1133 * 3.3.8.3.3 YUV420p Stream Combination for YUV444v2 mode */
1134 /* U: multiplications with subtotals and horizontal sums */
1135 __m128i ue;
1136 __m128i uo;
1137 __m128i uavg;
1138 {
1139 const __m128i u_factors = BGRX_U_FACTORS;
1140 const __m128i ue1 =
1141 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe1, u_factors),
1142 _mm_maddubs_epi16(xe2, u_factors)),
1143 U_SHIFT);
1144 const __m128i ue2 =
1145 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe3, u_factors),
1146 _mm_maddubs_epi16(xe4, u_factors)),
1147 U_SHIFT);
1148 const __m128i ueavg = _mm_hadd_epi16(ue1, ue2);
1149 ue = _mm_sub_epi8(_mm_packs_epi16(ue1, ue2), vector128);
1150 uavg = ueavg;
1151 }
1152 {
1153 const __m128i u_factors = BGRX_U_FACTORS;
1154 const __m128i uo1 =
1155 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo1, u_factors),
1156 _mm_maddubs_epi16(xo2, u_factors)),
1157 U_SHIFT);
1158 const __m128i uo2 =
1159 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo3, u_factors),
1160 _mm_maddubs_epi16(xo4, u_factors)),
1161 U_SHIFT);
1162 const __m128i uoavg = _mm_hadd_epi16(uo1, uo2);
1163 uo = _mm_sub_epi8(_mm_packs_epi16(uo1, uo2), vector128);
1164 uavg = _mm_add_epi16(uavg, uoavg);
1165 uavg = _mm_srai_epi16(uavg, 2);
1166 uavg = _mm_packs_epi16(uavg, uoavg);
1167 uavg = _mm_sub_epi8(uavg, vector128);
1168 }
1169 /* Now we need the following storage distribution:
1170 * 2x 2y -> uLumaDst
1171 * 2x+1 y -> yChromaDst1
1172 * 4x 2y+1 -> uChromaDst1
1173 * 4x+2 2y+1 -> vChromaDst1 */
1174 {
1175 const __m128i mask =
1176 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
1177 (char)0x80, (char)0x80, (char)0x80, 15, 13, 11, 9, 7, 5, 3, 1);
1178 const __m128i ude = _mm_shuffle_epi8(ue, mask);
1179 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, yEvenChromaDst1), ude);
1180 yEvenChromaDst1 += 8;
1181 }
1182
1183 if (yLumaDstOdd)
1184 {
1185 const __m128i mask =
1186 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
1187 (char)0x80, (char)0x80, (char)0x80, 15, 13, 11, 9, 7, 5, 3, 1);
1188 const __m128i udo /* codespell:ignore udo */ = _mm_shuffle_epi8(uo, mask);
1189 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, yOddChromaDst1),
1190 udo); // codespell:ignore udo
1191 yOddChromaDst1 += 8;
1192 }
1193
1194 if (yLumaDstOdd)
1195 {
1196 const __m128i mask =
1197 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
1198 (char)0x80, (char)0x80, (char)0x80, 14, 10, 6, 2, 12, 8, 4, 0);
1199 const __m128i ud = _mm_shuffle_epi8(uo, mask);
1200 int* uDst1 = WINPR_PACKED_ALIGN_CAST(int*, uChromaDst1);
1201 int* vDst1 = WINPR_PACKED_ALIGN_CAST(int*, vChromaDst1);
1202 const int* src = (const int*)&ud;
1203 _mm_stream_si32(uDst1, src[0]);
1204 _mm_stream_si32(vDst1, src[1]);
1205 uChromaDst1 += 4;
1206 vChromaDst1 += 4;
1207 }
1208
1209 if (yLumaDstOdd)
1210 {
1211 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, uLumaDst), uavg);
1212 uLumaDst += 8;
1213 }
1214 else
1215 {
1216 const __m128i mask =
1217 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
1218 (char)0x80, (char)0x80, (char)0x80, 14, 12, 10, 8, 6, 4, 2, 0);
1219 const __m128i ud = _mm_shuffle_epi8(ue, mask);
1220 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, uLumaDst), ud);
1221 uLumaDst += 8;
1222 }
1223 }
1224
1225 {
1226 /* V: multiplications with subtotals and horizontal sums */
1227 __m128i ve;
1228 __m128i vo;
1229 __m128i vavg;
1230 {
1231 const __m128i v_factors = BGRX_V_FACTORS;
1232 const __m128i ve1 =
1233 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe1, v_factors),
1234 _mm_maddubs_epi16(xe2, v_factors)),
1235 V_SHIFT);
1236 const __m128i ve2 =
1237 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xe3, v_factors),
1238 _mm_maddubs_epi16(xe4, v_factors)),
1239 V_SHIFT);
1240 const __m128i veavg = _mm_hadd_epi16(ve1, ve2);
1241 ve = _mm_sub_epi8(_mm_packs_epi16(ve1, ve2), vector128);
1242 vavg = veavg;
1243 }
1244 {
1245 const __m128i v_factors = BGRX_V_FACTORS;
1246 const __m128i vo1 =
1247 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo1, v_factors),
1248 _mm_maddubs_epi16(xo2, v_factors)),
1249 V_SHIFT);
1250 const __m128i vo2 =
1251 _mm_srai_epi16(_mm_hadd_epi16(_mm_maddubs_epi16(xo3, v_factors),
1252 _mm_maddubs_epi16(xo4, v_factors)),
1253 V_SHIFT);
1254 const __m128i voavg = _mm_hadd_epi16(vo1, vo2);
1255 vo = _mm_sub_epi8(_mm_packs_epi16(vo1, vo2), vector128);
1256 vavg = _mm_add_epi16(vavg, voavg);
1257 vavg = _mm_srai_epi16(vavg, 2);
1258 vavg = _mm_packs_epi16(vavg, voavg);
1259 vavg = _mm_sub_epi8(vavg, vector128);
1260 }
1261 /* Now we need the following storage distribution:
1262 * 2x 2y -> vLumaDst
1263 * 2x+1 y -> yChromaDst2
1264 * 4x 2y+1 -> uChromaDst2
1265 * 4x+2 2y+1 -> vChromaDst2 */
1266 {
1267 const __m128i mask =
1268 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
1269 (char)0x80, (char)0x80, (char)0x80, 15, 13, 11, 9, 7, 5, 3, 1);
1270 __m128i vde = _mm_shuffle_epi8(ve, mask);
1271 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, yEvenChromaDst2), vde);
1272 yEvenChromaDst2 += 8;
1273 }
1274
1275 if (yLumaDstOdd)
1276 {
1277 const __m128i mask =
1278 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
1279 (char)0x80, (char)0x80, (char)0x80, 15, 13, 11, 9, 7, 5, 3, 1);
1280 __m128i vdo = _mm_shuffle_epi8(vo, mask);
1281 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, yOddChromaDst2), vdo);
1282 yOddChromaDst2 += 8;
1283 }
1284
1285 if (yLumaDstOdd)
1286 {
1287 const __m128i mask =
1288 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
1289 (char)0x80, (char)0x80, (char)0x80, 14, 10, 6, 2, 12, 8, 4, 0);
1290 const __m128i vd = _mm_shuffle_epi8(vo, mask);
1291 int* uDst2 = WINPR_PACKED_ALIGN_CAST(int*, uChromaDst2);
1292 int* vDst2 = WINPR_PACKED_ALIGN_CAST(int*, vChromaDst2);
1293 const int* src = (const int*)&vd;
1294 _mm_stream_si32(uDst2, src[0]);
1295 _mm_stream_si32(vDst2, src[1]);
1296 uChromaDst2 += 4;
1297 vChromaDst2 += 4;
1298 }
1299
1300 if (yLumaDstOdd)
1301 {
1302 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, vLumaDst), vavg);
1303 vLumaDst += 8;
1304 }
1305 else
1306 {
1307 const __m128i mask =
1308 _mm_set_epi8((char)0x80, (char)0x80, (char)0x80, (char)0x80, (char)0x80,
1309 (char)0x80, (char)0x80, (char)0x80, 14, 12, 10, 8, 6, 4, 2, 0);
1310 __m128i vd = _mm_shuffle_epi8(ve, mask);
1311 _mm_storel_epi64(WINPR_PACKED_ALIGN_CAST(__m128i*, vLumaDst), vd);
1312 vLumaDst += 8;
1313 }
1314 }
1315 }
1316
1317 general_RGBToAVC444YUVv2_BGRX_DOUBLE_ROW(x, srcEven, srcOdd, yLumaDstEven, yLumaDstOdd,
1318 uLumaDst, vLumaDst, yEvenChromaDst1, yEvenChromaDst2,
1319 yOddChromaDst1, yOddChromaDst2, uChromaDst1,
1320 uChromaDst2, vChromaDst1, vChromaDst2, width);
1321}
1322
1323static pstatus_t sse41_RGBToAVC444YUVv2_BGRX(const BYTE* WINPR_RESTRICT pSrc,
1324 WINPR_ATTR_UNUSED UINT32 srcFormat, UINT32 srcStep,
1325 BYTE* WINPR_RESTRICT pDst1[], const UINT32 dst1Step[],
1326 BYTE* WINPR_RESTRICT pDst2[], const UINT32 dst2Step[],
1327 const prim_size_t* WINPR_RESTRICT roi)
1328{
1329 if (roi->height < 1 || roi->width < 1)
1330 return !PRIMITIVES_SUCCESS;
1331
1332 size_t y = 0;
1333 for (; y < roi->height - roi->height % 2; y += 2)
1334 {
1335 const BYTE* srcEven = (pSrc + y * srcStep);
1336 const BYTE* srcOdd = (srcEven + srcStep);
1337 BYTE* dstLumaYEven = (pDst1[0] + y * dst1Step[0]);
1338 BYTE* dstLumaYOdd = (dstLumaYEven + dst1Step[0]);
1339 BYTE* dstLumaU = (pDst1[1] + (y / 2) * dst1Step[1]);
1340 BYTE* dstLumaV = (pDst1[2] + (y / 2) * dst1Step[2]);
1341 BYTE* dstEvenChromaY1 = (pDst2[0] + y * dst2Step[0]);
1342 BYTE* dstEvenChromaY2 = dstEvenChromaY1 + roi->width / 2;
1343 BYTE* dstOddChromaY1 = dstEvenChromaY1 + dst2Step[0];
1344 BYTE* dstOddChromaY2 = dstEvenChromaY2 + dst2Step[0];
1345 BYTE* dstChromaU1 = (pDst2[1] + (y / 2) * dst2Step[1]);
1346 BYTE* dstChromaV1 = (pDst2[2] + (y / 2) * dst2Step[2]);
1347 BYTE* dstChromaU2 = dstChromaU1 + roi->width / 4;
1348 BYTE* dstChromaV2 = dstChromaV1 + roi->width / 4;
1349 sse41_RGBToAVC444YUVv2_BGRX_DOUBLE_ROW(srcEven, srcOdd, dstLumaYEven, dstLumaYOdd, dstLumaU,
1350 dstLumaV, dstEvenChromaY1, dstEvenChromaY2,
1351 dstOddChromaY1, dstOddChromaY2, dstChromaU1,
1352 dstChromaU2, dstChromaV1, dstChromaV2, roi->width);
1353 }
1354
1355 for (; y < roi->height; y++)
1356 {
1357 const BYTE* srcEven = (pSrc + y * srcStep);
1358 BYTE* dstLumaYEven = (pDst1[0] + y * dst1Step[0]);
1359 BYTE* dstLumaU = (pDst1[1] + (y / 2) * dst1Step[1]);
1360 BYTE* dstLumaV = (pDst1[2] + (y / 2) * dst1Step[2]);
1361 BYTE* dstEvenChromaY1 = (pDst2[0] + y * dst2Step[0]);
1362 BYTE* dstEvenChromaY2 = dstEvenChromaY1 + roi->width / 2;
1363 BYTE* dstChromaU1 = (pDst2[1] + (y / 2) * dst2Step[1]);
1364 BYTE* dstChromaV1 = (pDst2[2] + (y / 2) * dst2Step[2]);
1365 BYTE* dstChromaU2 = dstChromaU1 + roi->width / 4;
1366 BYTE* dstChromaV2 = dstChromaV1 + roi->width / 4;
1367 general_RGBToAVC444YUVv2_BGRX_DOUBLE_ROW(0, srcEven, nullptr, dstLumaYEven, nullptr,
1368 dstLumaU, dstLumaV, dstEvenChromaY1,
1369 dstEvenChromaY2, nullptr, nullptr, dstChromaU1,
1370 dstChromaU2, dstChromaV1, dstChromaV2, roi->width);
1371 }
1372
1373 return PRIMITIVES_SUCCESS;
1374}
1375
1376static pstatus_t sse41_RGBToAVC444YUVv2(const BYTE* WINPR_RESTRICT pSrc, UINT32 srcFormat,
1377 UINT32 srcStep, BYTE* WINPR_RESTRICT pDst1[],
1378 const UINT32 dst1Step[], BYTE* WINPR_RESTRICT pDst2[],
1379 const UINT32 dst2Step[],
1380 const prim_size_t* WINPR_RESTRICT roi)
1381{
1382 switch (srcFormat)
1383 {
1384 case PIXEL_FORMAT_BGRX32:
1385 case PIXEL_FORMAT_BGRA32:
1386 return sse41_RGBToAVC444YUVv2_BGRX(pSrc, srcFormat, srcStep, pDst1, dst1Step, pDst2,
1387 dst2Step, roi);
1388
1389 default:
1390 return generic->RGBToAVC444YUVv2(pSrc, srcFormat, srcStep, pDst1, dst1Step, pDst2,
1391 dst2Step, roi);
1392 }
1393}
1394
1395static pstatus_t sse41_LumaToYUV444(const BYTE* WINPR_RESTRICT pSrcRaw[], const UINT32 srcStep[],
1396 BYTE* WINPR_RESTRICT pDstRaw[], const UINT32 dstStep[],
1397 const RECTANGLE_16* WINPR_RESTRICT roi)
1398{
1399 const UINT32 nWidth = roi->right - roi->left;
1400 const UINT32 nHeight = roi->bottom - roi->top;
1401 const UINT32 halfWidth = (nWidth + 1) / 2;
1402 const UINT32 halfPad = halfWidth % 16;
1403 const UINT32 halfHeight = (nHeight + 1) / 2;
1404 const UINT32 oddY = 1;
1405 const UINT32 evenY = 0;
1406 const UINT32 oddX = 1;
1407 const UINT32 evenX = 0;
1408 const BYTE* pSrc[3] = { pSrcRaw[0] + 1ULL * roi->top * srcStep[0] + roi->left,
1409 pSrcRaw[1] + 1ULL * roi->top / 2 * srcStep[1] + roi->left / 2,
1410 pSrcRaw[2] + 1ULL * roi->top / 2 * srcStep[2] + roi->left / 2 };
1411 BYTE* pDst[3] = { pDstRaw[0] + 1ULL * roi->top * dstStep[0] + roi->left,
1412 pDstRaw[1] + 1ULL * roi->top * dstStep[1] + roi->left,
1413 pDstRaw[2] + 1ULL * roi->top * dstStep[2] + roi->left };
1414
1415 /* Y data is already here... */
1416 /* B1 */
1417 for (size_t y = 0; y < nHeight; y++)
1418 {
1419 const BYTE* Ym = pSrc[0] + y * srcStep[0];
1420 BYTE* pY = pDst[0] + y * dstStep[0];
1421 memcpy(pY, Ym, nWidth);
1422 }
1423
1424 /* The first half of U, V are already here part of this frame. */
1425 /* B2 and B3 */
1426 for (size_t y = 0; y < halfHeight; y++)
1427 {
1428 const size_t val2y = (2 * y + evenY);
1429 const size_t val2y1 = val2y + oddY;
1430 const BYTE* Um = pSrc[1] + 1ULL * srcStep[1] * y;
1431 const BYTE* Vm = pSrc[2] + 1ULL * srcStep[2] * y;
1432 BYTE* pU = pDst[1] + 1ULL * dstStep[1] * val2y;
1433 BYTE* pV = pDst[2] + 1ULL * dstStep[2] * val2y;
1434 BYTE* pU1 = pDst[1] + 1ULL * dstStep[1] * val2y1;
1435 BYTE* pV1 = pDst[2] + 1ULL * dstStep[2] * val2y1;
1436
1437 size_t x = 0;
1438 for (; x < halfWidth - halfPad; x += 16)
1439 {
1440 const __m128i unpackHigh = _mm_set_epi8(7, 7, 6, 6, 5, 5, 4, 4, 3, 3, 2, 2, 1, 1, 0, 0);
1441 const __m128i unpackLow =
1442 _mm_set_epi8(15, 15, 14, 14, 13, 13, 12, 12, 11, 11, 10, 10, 9, 9, 8, 8);
1443 {
1444 const __m128i u = LOAD_SI128(&Um[x]);
1445 const __m128i uHigh = _mm_shuffle_epi8(u, unpackHigh);
1446 const __m128i uLow = _mm_shuffle_epi8(u, unpackLow);
1447 STORE_SI128(&pU[2ULL * x], uHigh);
1448 STORE_SI128(&pU[2ULL * x + 16], uLow);
1449 STORE_SI128(&pU1[2ULL * x], uHigh);
1450 STORE_SI128(&pU1[2ULL * x + 16], uLow);
1451 }
1452 {
1453 const __m128i u = LOAD_SI128(&Vm[x]);
1454 const __m128i uHigh = _mm_shuffle_epi8(u, unpackHigh);
1455 const __m128i uLow = _mm_shuffle_epi8(u, unpackLow);
1456 STORE_SI128(&pV[2 * x], uHigh);
1457 STORE_SI128(&pV[2 * x + 16], uLow);
1458 STORE_SI128(&pV1[2 * x], uHigh);
1459 STORE_SI128(&pV1[2 * x + 16], uLow);
1460 }
1461 }
1462
1463 for (; x < halfWidth; x++)
1464 {
1465 const size_t val2x = 2 * x + evenX;
1466 const size_t val2x1 = val2x + oddX;
1467 pU[val2x] = Um[x];
1468 pV[val2x] = Vm[x];
1469 pU[val2x1] = Um[x];
1470 pV[val2x1] = Vm[x];
1471 pU1[val2x] = Um[x];
1472 pV1[val2x] = Vm[x];
1473 pU1[val2x1] = Um[x];
1474 pV1[val2x1] = Vm[x];
1475 }
1476 }
1477
1478 return PRIMITIVES_SUCCESS;
1479}
1480
1481static pstatus_t sse41_ChromaV1ToYUV444(const BYTE* WINPR_RESTRICT pSrcRaw[3],
1482 const UINT32 srcStep[3], BYTE* WINPR_RESTRICT pDstRaw[3],
1483 const UINT32 dstStep[3],
1484 const RECTANGLE_16* WINPR_RESTRICT roi)
1485{
1486 const UINT32 mod = 16;
1487 UINT32 uY = 0;
1488 UINT32 vY = 0;
1489 const UINT32 nWidth = roi->right - roi->left;
1490 const UINT32 nHeight = roi->bottom - roi->top;
1491 const UINT32 halfWidth = (nWidth + 1) / 2;
1492 const UINT32 halfPad = halfWidth % 16;
1493 const UINT32 halfHeight = (nHeight + 1) / 2;
1494 const UINT32 oddY = 1;
1495 const UINT32 evenY = 0;
1496 const UINT32 oddX = 1;
1497 /* The auxiliary frame is aligned to multiples of 16x16.
1498 * We need the padded height for B4 and B5 conversion. */
1499 const UINT32 padHeight = nHeight + 16 - nHeight % 16;
1500 const BYTE* pSrc[3] = { pSrcRaw[0] + 1ULL * roi->top * srcStep[0] + roi->left,
1501 pSrcRaw[1] + 1ULL * roi->top / 2 * srcStep[1] + roi->left / 2,
1502 pSrcRaw[2] + 1ULL * roi->top / 2 * srcStep[2] + roi->left / 2 };
1503 BYTE* pDst[3] = { pDstRaw[0] + 1ULL * roi->top * dstStep[0] + roi->left,
1504 pDstRaw[1] + 1ULL * roi->top * dstStep[1] + roi->left,
1505 pDstRaw[2] + 1ULL * roi->top * dstStep[2] + roi->left };
1506 const __m128i zero = _mm_setzero_si128();
1507 const __m128i mask = _mm_set_epi8(0, (char)0x80, 0, (char)0x80, 0, (char)0x80, 0, (char)0x80, 0,
1508 (char)0x80, 0, (char)0x80, 0, (char)0x80, 0, (char)0x80);
1509
1510 /* The second half of U and V is a bit more tricky... */
1511 /* B4 and B5 */
1512 for (size_t y = 0; y < padHeight; y++)
1513 {
1514 const BYTE* Ya = pSrc[0] + 1ULL * srcStep[0] * y;
1515 BYTE* pX = nullptr;
1516
1517 if ((y) % mod < (mod + 1) / 2)
1518 {
1519 const UINT32 pos = (2 * uY++ + oddY);
1520
1521 if (pos >= nHeight)
1522 continue;
1523
1524 pX = pDst[1] + 1ULL * dstStep[1] * pos;
1525 }
1526 else
1527 {
1528 const UINT32 pos = (2 * vY++ + oddY);
1529
1530 if (pos >= nHeight)
1531 continue;
1532
1533 pX = pDst[2] + 1ULL * dstStep[2] * pos;
1534 }
1535
1536 if (y < nHeight)
1537 memcpy(pX, Ya, nWidth);
1538 }
1539
1540 /* B6 and B7 */
1541 for (size_t y = 0; y < halfHeight; y++)
1542 {
1543 const size_t val2y = (y * 2 + evenY);
1544 const BYTE* Ua = pSrc[1] + srcStep[1] * y;
1545 const BYTE* Va = pSrc[2] + srcStep[2] * y;
1546 BYTE* pU = pDst[1] + dstStep[1] * val2y;
1547 BYTE* pV = pDst[2] + dstStep[2] * val2y;
1548
1549 size_t x = 0;
1550 for (; x < halfWidth - halfPad; x += 16)
1551 {
1552 {
1553 const __m128i u = LOAD_SI128(&Ua[x]);
1554 const __m128i u2 = _mm_unpackhi_epi8(u, zero);
1555 const __m128i u1 = _mm_unpacklo_epi8(u, zero);
1556 _mm_maskmoveu_si128(u1, mask, (char*)&pU[2 * x]);
1557 _mm_maskmoveu_si128(u2, mask, (char*)&pU[2 * x + 16]);
1558 }
1559 {
1560 const __m128i u = LOAD_SI128(&Va[x]);
1561 const __m128i u2 = _mm_unpackhi_epi8(u, zero);
1562 const __m128i u1 = _mm_unpacklo_epi8(u, zero);
1563 _mm_maskmoveu_si128(u1, mask, (char*)&pV[2 * x]);
1564 _mm_maskmoveu_si128(u2, mask, (char*)&pV[2 * x + 16]);
1565 }
1566 }
1567
1568 for (; x < halfWidth; x++)
1569 {
1570 const size_t val2x1 = (x * 2ULL + oddX);
1571 pU[val2x1] = Ua[x];
1572 pV[val2x1] = Va[x];
1573 }
1574 }
1575
1576 return PRIMITIVES_SUCCESS;
1577}
1578
1579static pstatus_t sse41_ChromaV2ToYUV444(const BYTE* WINPR_RESTRICT pSrc[3], const UINT32 srcStep[3],
1580 UINT32 nTotalWidth, WINPR_ATTR_UNUSED UINT32 nTotalHeight,
1581 BYTE* WINPR_RESTRICT pDst[3], const UINT32 dstStep[3],
1582 const RECTANGLE_16* WINPR_RESTRICT roi)
1583{
1584 const UINT32 nWidth = roi->right - roi->left;
1585 const UINT32 nHeight = roi->bottom - roi->top;
1586 const UINT32 halfWidth = (nWidth + 1) / 2;
1587 const UINT32 halfPad = halfWidth % 16;
1588 const UINT32 halfHeight = (nHeight + 1) / 2;
1589 const UINT32 quaterWidth = (nWidth + 3) / 4;
1590 const UINT32 quaterPad = quaterWidth % 16;
1591 const __m128i zero = _mm_setzero_si128();
1592 const __m128i mask = _mm_set_epi8((char)0x80, 0, (char)0x80, 0, (char)0x80, 0, (char)0x80, 0,
1593 (char)0x80, 0, (char)0x80, 0, (char)0x80, 0, (char)0x80, 0);
1594 const __m128i mask2 = _mm_set_epi8(0, (char)0x80, 0, (char)0x80, 0, (char)0x80, 0, (char)0x80,
1595 0, (char)0x80, 0, (char)0x80, 0, (char)0x80, 0, (char)0x80);
1596 const __m128i shuffle1 =
1597 _mm_set_epi8((char)0x80, 15, (char)0x80, 14, (char)0x80, 13, (char)0x80, 12, (char)0x80, 11,
1598 (char)0x80, 10, (char)0x80, 9, (char)0x80, 8);
1599 const __m128i shuffle2 =
1600 _mm_set_epi8((char)0x80, 7, (char)0x80, 6, (char)0x80, 5, (char)0x80, 4, (char)0x80, 3,
1601 (char)0x80, 2, (char)0x80, 1, (char)0x80, 0);
1602
1603 /* B4 and B5: odd UV values for width/2, height */
1604 for (size_t y = 0; y < nHeight; y++)
1605 {
1606 const size_t yTop = y + roi->top;
1607 const BYTE* pYaU = pSrc[0] + srcStep[0] * yTop + roi->left / 2;
1608 const BYTE* pYaV = pYaU + nTotalWidth / 2;
1609 BYTE* pU = pDst[1] + 1ULL * dstStep[1] * yTop + roi->left;
1610 BYTE* pV = pDst[2] + 1ULL * dstStep[2] * yTop + roi->left;
1611
1612 size_t x = 0;
1613 for (; x < halfWidth - halfPad; x += 16)
1614 {
1615 {
1616 const __m128i u = LOAD_SI128(&pYaU[x]);
1617 const __m128i u2 = _mm_unpackhi_epi8(zero, u);
1618 const __m128i u1 = _mm_unpacklo_epi8(zero, u);
1619 _mm_maskmoveu_si128(u1, mask, (char*)&pU[2 * x]);
1620 _mm_maskmoveu_si128(u2, mask, (char*)&pU[2 * x + 16]);
1621 }
1622 {
1623 const __m128i v = LOAD_SI128(&pYaV[x]);
1624 const __m128i v2 = _mm_unpackhi_epi8(zero, v);
1625 const __m128i v1 = _mm_unpacklo_epi8(zero, v);
1626 _mm_maskmoveu_si128(v1, mask, (char*)&pV[2 * x]);
1627 _mm_maskmoveu_si128(v2, mask, (char*)&pV[2 * x + 16]);
1628 }
1629 }
1630
1631 for (; x < halfWidth; x++)
1632 {
1633 const size_t odd = 2ULL * x + 1;
1634 pU[odd] = pYaU[x];
1635 pV[odd] = pYaV[x];
1636 }
1637 }
1638
1639 /* B6 - B9 */
1640 for (size_t y = 0; y < halfHeight; y++)
1641 {
1642 const BYTE* pUaU = pSrc[1] + srcStep[1] * (y + roi->top / 2) + roi->left / 4;
1643 const BYTE* pUaV = pUaU + nTotalWidth / 4;
1644 const BYTE* pVaU = pSrc[2] + srcStep[2] * (y + roi->top / 2) + roi->left / 4;
1645 const BYTE* pVaV = pVaU + nTotalWidth / 4;
1646 BYTE* pU = pDst[1] + dstStep[1] * (2 * y + 1 + roi->top) + roi->left;
1647 BYTE* pV = pDst[2] + dstStep[2] * (2 * y + 1 + roi->top) + roi->left;
1648
1649 UINT32 x = 0;
1650 for (; x < quaterWidth - quaterPad; x += 16)
1651 {
1652 {
1653 const __m128i uU = LOAD_SI128(&pUaU[x]);
1654 const __m128i uV = LOAD_SI128(&pVaU[x]);
1655 const __m128i uHigh = _mm_unpackhi_epi8(uU, uV);
1656 const __m128i uLow = _mm_unpacklo_epi8(uU, uV);
1657 const __m128i u1 = _mm_shuffle_epi8(uLow, shuffle2);
1658 const __m128i u2 = _mm_shuffle_epi8(uLow, shuffle1);
1659 const __m128i u3 = _mm_shuffle_epi8(uHigh, shuffle2);
1660 const __m128i u4 = _mm_shuffle_epi8(uHigh, shuffle1);
1661 _mm_maskmoveu_si128(u1, mask2, (char*)&pU[4 * x + 0]);
1662 _mm_maskmoveu_si128(u2, mask2, (char*)&pU[4 * x + 16]);
1663 _mm_maskmoveu_si128(u3, mask2, (char*)&pU[4 * x + 32]);
1664 _mm_maskmoveu_si128(u4, mask2, (char*)&pU[4 * x + 48]);
1665 }
1666 {
1667 const __m128i vU = LOAD_SI128(&pUaV[x]);
1668 const __m128i vV = LOAD_SI128(&pVaV[x]);
1669 const __m128i vHigh = _mm_unpackhi_epi8(vU, vV);
1670 const __m128i vLow = _mm_unpacklo_epi8(vU, vV);
1671 const __m128i v1 = _mm_shuffle_epi8(vLow, shuffle2);
1672 const __m128i v2 = _mm_shuffle_epi8(vLow, shuffle1);
1673 const __m128i v3 = _mm_shuffle_epi8(vHigh, shuffle2);
1674 const __m128i v4 = _mm_shuffle_epi8(vHigh, shuffle1);
1675 _mm_maskmoveu_si128(v1, mask2, (char*)&pV[4 * x + 0]);
1676 _mm_maskmoveu_si128(v2, mask2, (char*)&pV[4 * x + 16]);
1677 _mm_maskmoveu_si128(v3, mask2, (char*)&pV[4 * x + 32]);
1678 _mm_maskmoveu_si128(v4, mask2, (char*)&pV[4 * x + 48]);
1679 }
1680 }
1681
1682 for (; x < quaterWidth; x++)
1683 {
1684 pU[4 * x + 0] = pUaU[x];
1685 pV[4 * x + 0] = pUaV[x];
1686 pU[4 * x + 2] = pVaU[x];
1687 pV[4 * x + 2] = pVaV[x];
1688 }
1689 }
1690
1691 return PRIMITIVES_SUCCESS;
1692}
1693
1694static pstatus_t sse41_YUV420CombineToYUV444(avc444_frame_type type,
1695 const BYTE* WINPR_RESTRICT pSrc[3],
1696 const UINT32 srcStep[3], UINT32 nWidth, UINT32 nHeight,
1697 BYTE* WINPR_RESTRICT pDst[3], const UINT32 dstStep[3],
1698 const RECTANGLE_16* WINPR_RESTRICT roi)
1699{
1700 if (!pSrc || !pSrc[0] || !pSrc[1] || !pSrc[2])
1701 return -1;
1702
1703 if (!pDst || !pDst[0] || !pDst[1] || !pDst[2])
1704 return -1;
1705
1706 if (!roi)
1707 return -1;
1708
1709 switch (type)
1710 {
1711 case AVC444_LUMA:
1712 return sse41_LumaToYUV444(pSrc, srcStep, pDst, dstStep, roi);
1713
1714 case AVC444_CHROMAv1:
1715 return sse41_ChromaV1ToYUV444(pSrc, srcStep, pDst, dstStep, roi);
1716
1717 case AVC444_CHROMAv2:
1718 return sse41_ChromaV2ToYUV444(pSrc, srcStep, nWidth, nHeight, pDst, dstStep, roi);
1719
1720 default:
1721 return -1;
1722 }
1723}
1724#endif
1725
1726void primitives_init_YUV_sse41_int(primitives_t* WINPR_RESTRICT prims)
1727{
1728#if defined(SSE_AVX_INTRINSICS_ENABLED)
1729 generic = primitives_get_generic();
1730
1731 WLog_VRB(PRIM_TAG, "SSE3/sse41 optimizations");
1732 prims->RGBToYUV420_8u_P3AC4R = sse41_RGBToYUV420;
1733 prims->RGBToAVC444YUV = sse41_RGBToAVC444YUV;
1734 prims->RGBToAVC444YUVv2 = sse41_RGBToAVC444YUVv2;
1735 prims->YUV420ToRGB_8u_P3AC4R = sse41_YUV420ToRGB;
1736 prims->YUV444ToRGB_8u_P3AC4R = sse41_YUV444ToRGB_8u_P3AC4R;
1737 prims->YUV420CombineToYUV444 = sse41_YUV420CombineToYUV444;
1738#else
1739 WLog_VRB(PRIM_TAG, "undefined WITH_SIMD or sse41 intrinsics not available");
1740 WINPR_UNUSED(prims);
1741#endif
1742}