+ if (((height) % 2) == 0) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Dc(i) -= OPJ_Sc(i);
+ }
+ }
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Sc(0) += (OPJ_Dc(0) + OPJ_Dc(0) + 2) >> 2;
+ }
+ i = 1;
+ if (i < dn) {
+ __m128i xmm_Dim1_0 = *(const __m128i*)(tmp + (1 +
+ (i - 1) * 2) * NB_ELTS_V8 + 4 * 0);
+ __m128i xmm_Dim1_1 = *(const __m128i*)(tmp + (1 +
+ (i - 1) * 2) * NB_ELTS_V8 + 4 * 1);
+ const __m128i xmm_two = _mm_set1_epi32(2);
+ for (; i < dn; i++) {
+ __m128i xmm_Di_0 = *(const __m128i*)(tmp +
+ (1 + i * 2) * NB_ELTS_V8 + 4 * 0);
+ __m128i xmm_Di_1 = *(const __m128i*)(tmp +
+ (1 + i * 2) * NB_ELTS_V8 + 4 * 1);
+ __m128i xmm_Si_0 = *(const __m128i*)(tmp +
+ (i * 2) * NB_ELTS_V8 + 4 * 0);
+ __m128i xmm_Si_1 = *(const __m128i*)(tmp +
+ (i * 2) * NB_ELTS_V8 + 4 * 1);
+ xmm_Si_0 = _mm_add_epi32(xmm_Si_0,
+ _mm_srai_epi32(_mm_add_epi32(_mm_add_epi32(xmm_Dim1_0, xmm_Di_0), xmm_two), 2));
+ xmm_Si_1 = _mm_add_epi32(xmm_Si_1,
+ _mm_srai_epi32(_mm_add_epi32(_mm_add_epi32(xmm_Dim1_1, xmm_Di_1), xmm_two), 2));
+ *(__m128i*)(tmp + (i * 2) * NB_ELTS_V8 + 4 * 0) = xmm_Si_0;
+ *(__m128i*)(tmp + (i * 2) * NB_ELTS_V8 + 4 * 1) = xmm_Si_1;
+ xmm_Dim1_0 = xmm_Di_0;
+ xmm_Dim1_1 = xmm_Di_1;
+ }
+ }
+ if (((height) % 2) == 1) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Sc(i) += (OPJ_Dc(i - 1) + OPJ_Dc(i - 1) + 2) >> 2;
+ }
+ }
+ } else {
+ OPJ_UINT32 c;
+ OPJ_UINT32 i;
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Sc(0) -= OPJ_Dc(0);
+ }
+ i = 1;
+ if (i < sn) {
+ __m128i xmm_Dim1_0 = *(const __m128i*)(tmp + (1 +
+ (i - 1) * 2) * NB_ELTS_V8 + 4 * 0);
+ __m128i xmm_Dim1_1 = *(const __m128i*)(tmp + (1 +
+ (i - 1) * 2) * NB_ELTS_V8 + 4 * 1);
+ for (; i < sn; i++) {
+ __m128i xmm_Di_0 = *(const __m128i*)(tmp +
+ (1 + i * 2) * NB_ELTS_V8 + 4 * 0);
+ __m128i xmm_Di_1 = *(const __m128i*)(tmp +
+ (1 + i * 2) * NB_ELTS_V8 + 4 * 1);
+ __m128i xmm_Si_0 = *(const __m128i*)(tmp +
+ (i * 2) * NB_ELTS_V8 + 4 * 0);
+ __m128i xmm_Si_1 = *(const __m128i*)(tmp +
+ (i * 2) * NB_ELTS_V8 + 4 * 1);
+ xmm_Si_0 = _mm_sub_epi32(xmm_Si_0,
+ _mm_srai_epi32(_mm_add_epi32(xmm_Di_0, xmm_Dim1_0), 1));
+ xmm_Si_1 = _mm_sub_epi32(xmm_Si_1,
+ _mm_srai_epi32(_mm_add_epi32(xmm_Di_1, xmm_Dim1_1), 1));
+ *(__m128i*)(tmp + (i * 2) * NB_ELTS_V8 + 4 * 0) = xmm_Si_0;
+ *(__m128i*)(tmp + (i * 2) * NB_ELTS_V8 + 4 * 1) = xmm_Si_1;
+ xmm_Dim1_0 = xmm_Di_0;
+ xmm_Dim1_1 = xmm_Di_1;
+ }
+ }
+ if (((height) % 2) == 1) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Sc(i) -= OPJ_Dc(i - 1);
+ }
+ }
+ i = 0;
+ if (i + 1 < dn) {
+ __m128i xmm_Si_0 = *((const __m128i*)(tmp + 4 * 0));
+ __m128i xmm_Si_1 = *((const __m128i*)(tmp + 4 * 1));
+ const __m128i xmm_two = _mm_set1_epi32(2);
+ for (; i + 1 < dn; i++) {
+ __m128i xmm_Sip1_0 = *(const __m128i*)(tmp +
+ (i + 1) * 2 * NB_ELTS_V8 + 4 * 0);
+ __m128i xmm_Sip1_1 = *(const __m128i*)(tmp +
+ (i + 1) * 2 * NB_ELTS_V8 + 4 * 1);
+ __m128i xmm_Di_0 = *(const __m128i*)(tmp +
+ (1 + i * 2) * NB_ELTS_V8 + 4 * 0);
+ __m128i xmm_Di_1 = *(const __m128i*)(tmp +
+ (1 + i * 2) * NB_ELTS_V8 + 4 * 1);
+ xmm_Di_0 = _mm_add_epi32(xmm_Di_0,
+ _mm_srai_epi32(_mm_add_epi32(_mm_add_epi32(xmm_Si_0, xmm_Sip1_0), xmm_two), 2));
+ xmm_Di_1 = _mm_add_epi32(xmm_Di_1,
+ _mm_srai_epi32(_mm_add_epi32(_mm_add_epi32(xmm_Si_1, xmm_Sip1_1), xmm_two), 2));
+ *(__m128i*)(tmp + (1 + i * 2) * NB_ELTS_V8 + 4 * 0) = xmm_Di_0;
+ *(__m128i*)(tmp + (1 + i * 2) * NB_ELTS_V8 + 4 * 1) = xmm_Di_1;
+ xmm_Si_0 = xmm_Sip1_0;
+ xmm_Si_1 = xmm_Sip1_1;
+ }
+ }
+ if (((height) % 2) == 0) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Dc(i) += (OPJ_Sc(i) + OPJ_Sc(i) + 2) >> 2;
+ }
+ }
+ }
+#else
+ if (even) {
+ OPJ_UINT32 c;
+ if (height > 1) {
+ OPJ_UINT32 i;
+ for (i = 0; i + 1 < sn; i++) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Dc(i) -= (OPJ_Sc(i) + OPJ_Sc(i + 1)) >> 1;
+ }
+ }
+ if (((height) % 2) == 0) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Dc(i) -= OPJ_Sc(i);
+ }
+ }
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Sc(0) += (OPJ_Dc(0) + OPJ_Dc(0) + 2) >> 2;
+ }
+ for (i = 1; i < dn; i++) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Sc(i) += (OPJ_Dc(i - 1) + OPJ_Dc(i) + 2) >> 2;
+ }
+ }
+ if (((height) % 2) == 1) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Sc(i) += (OPJ_Dc(i - 1) + OPJ_Dc(i - 1) + 2) >> 2;
+ }
+ }
+ }
+ } else {
+ OPJ_UINT32 c;
+ if (height == 1) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Sc(0) *= 2;
+ }
+ } else {
+ OPJ_UINT32 i;
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Sc(0) -= OPJ_Dc(0);
+ }
+ for (i = 1; i < sn; i++) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Sc(i) -= (OPJ_Dc(i) + OPJ_Dc(i - 1)) >> 1;
+ }
+ }
+ if (((height) % 2) == 1) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Sc(i) -= OPJ_Dc(i - 1);
+ }
+ }
+ for (i = 0; i + 1 < dn; i++) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Dc(i) += (OPJ_Sc(i) + OPJ_Sc(i + 1) + 2) >> 2;
+ }
+ }
+ if (((height) % 2) == 0) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ OPJ_Dc(i) += (OPJ_Sc(i) + OPJ_Sc(i) + 2) >> 2;
+ }
+ }
+ }
+ }
+#endif
+
+ if (cols == NB_ELTS_V8) {
+ opj_dwt_deinterleave_v_cols(tmp, array, (OPJ_INT32)dn, (OPJ_INT32)sn,
+ stride_width, even ? 0 : 1, NB_ELTS_V8);
+ } else {
+ opj_dwt_deinterleave_v_cols(tmp, array, (OPJ_INT32)dn, (OPJ_INT32)sn,
+ stride_width, even ? 0 : 1, cols);
+ }
+}
+
+static void opj_v8dwt_encode_step1(OPJ_FLOAT32* fw,
+ OPJ_UINT32 end,
+ const OPJ_FLOAT32 cst)
+{
+ OPJ_UINT32 i;
+#ifdef __SSE__
+ __m128* vw = (__m128*) fw;
+ const __m128 vcst = _mm_set1_ps(cst);
+ for (i = 0; i < end; ++i) {
+ vw[0] = _mm_mul_ps(vw[0], vcst);
+ vw[1] = _mm_mul_ps(vw[1], vcst);
+ vw += 2 * (NB_ELTS_V8 * sizeof(OPJ_FLOAT32) / sizeof(__m128));
+ }
+#else
+ OPJ_UINT32 c;
+ for (i = 0; i < end; ++i) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ fw[i * 2 * NB_ELTS_V8 + c] *= cst;
+ }
+ }
+#endif
+}
+
+static void opj_v8dwt_encode_step2(OPJ_FLOAT32* fl, OPJ_FLOAT32* fw,
+ OPJ_UINT32 end,
+ OPJ_UINT32 m,
+ OPJ_FLOAT32 cst)
+{
+ OPJ_UINT32 i;
+ OPJ_UINT32 imax = opj_uint_min(end, m);
+#ifdef __SSE__
+ __m128* vw = (__m128*) fw;
+ __m128 vcst = _mm_set1_ps(cst);
+ if (imax > 0) {
+ __m128* vl = (__m128*) fl;
+ vw[-2] = _mm_add_ps(vw[-2], _mm_mul_ps(_mm_add_ps(vl[0], vw[0]), vcst));
+ vw[-1] = _mm_add_ps(vw[-1], _mm_mul_ps(_mm_add_ps(vl[1], vw[1]), vcst));
+ vw += 2 * (NB_ELTS_V8 * sizeof(OPJ_FLOAT32) / sizeof(__m128));
+ i = 1;
+
+ for (; i < imax; ++i) {
+ vw[-2] = _mm_add_ps(vw[-2], _mm_mul_ps(_mm_add_ps(vw[-4], vw[0]), vcst));
+ vw[-1] = _mm_add_ps(vw[-1], _mm_mul_ps(_mm_add_ps(vw[-3], vw[1]), vcst));
+ vw += 2 * (NB_ELTS_V8 * sizeof(OPJ_FLOAT32) / sizeof(__m128));
+ }
+ }
+ if (m < end) {
+ assert(m + 1 == end);
+ vcst = _mm_add_ps(vcst, vcst);
+ vw[-2] = _mm_add_ps(vw[-2], _mm_mul_ps(vw[-4], vcst));
+ vw[-1] = _mm_add_ps(vw[-1], _mm_mul_ps(vw[-3], vcst));
+ }
+#else
+ OPJ_INT32 c;
+ if (imax > 0) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ fw[-1 * NB_ELTS_V8 + c] += (fl[0 * NB_ELTS_V8 + c] + fw[0 * NB_ELTS_V8 + c]) *
+ cst;
+ }
+ fw += 2 * NB_ELTS_V8;
+ i = 1;
+ for (; i < imax; ++i) {
+ for (c = 0; c < NB_ELTS_V8; c++) {
+ fw[-1 * NB_ELTS_V8 + c] += (fw[-2 * NB_ELTS_V8 + c] + fw[0 * NB_ELTS_V8 + c]) *
+ cst;