1027 row = vsx_i64x2_const(0, 0);
1028 w0 = vsx_i8x16_swizzle(inf_u_q,
1029 vsx_i16x8_const(0x0100, 0x0100, 0x0100, 0x0100,
1030 0x0504, 0x0504, 0x0504, 0x0504));
1032 flags = vsx_v128_and(w0,
1033 vsx_u16x8_const(0x1110, 0x2220, 0x4440, 0x8880,
1034 0x1110, 0x2220, 0x4440, 0x8880));
1035 insig = vsx_i16x8_eq(flags, vsx_i64x2_const(0, 0));
1036 if (vsx_i8x16_bitmask(insig) != 0xFFFF)
1038 U_q = vsx_i8x16_swizzle(U_q,
1039 vsx_i16x8_const(0x0100, 0x0100, 0x0100, 0x0100,
1040 0x0504, 0x0504, 0x0504, 0x0504));
1041 flags = vsx_i16x8_mul(flags, vsx_i16x8_const(8,4,2,1,8,4,2,1));
1050 w0 = vsx_u16x8_shr(flags, 15);
1051 m_n = vsx_i16x8_sub(U_q, w0);
1052 m_n = vsx_v128_andnot(m_n, insig);
1056 v128_t ex_sum, shfl, inc_sum = m_n;
1058 inc_sum, 7, 8, 9, 10, 11, 12, 13, 14);
1059 inc_sum = vsx_i16x8_add(inc_sum, shfl);
1061 inc_sum = vsx_i16x8_add(inc_sum, shfl);
1063 inc_sum = vsx_i16x8_add(inc_sum, shfl);
1066 inc_sum, 7, 8, 9, 10, 11, 12, 13, 14);
1069 v128_t byte_idx = vsx_u16x8_shr(ex_sum, 3);
1071 vsx_v128_and(ex_sum, vsx_i16x8_const(
OJPH_REPEAT8(7)));
1072 byte_idx = vsx_i8x16_swizzle(byte_idx,
1073 vsx_i16x8_const(0x0000, 0x0202, 0x0404, 0x0606,
1074 0x0808, 0x0A0A, 0x0C0C, 0x0E0E));
1076 vsx_i16x8_add(byte_idx, vsx_i16x8_const(
OJPH_REPEAT8(0x0100)));
1077 v128_t d0 = vsx_i8x16_swizzle(ms_vec, byte_idx);
1079 vsx_i16x8_add(byte_idx, vsx_i16x8_const(
OJPH_REPEAT8(0x0101)));
1080 v128_t d1 = vsx_i8x16_swizzle(ms_vec, byte_idx);
1083 v128_t bit_shift = vsx_i8x16_swizzle(
1084 vsx_i8x16_const(-1, 127, 63, 31, 15, 7, 3, 1,
1085 -1, 127, 63, 31, 15, 7, 3, 1), bit_idx);
1087 vsx_i16x8_add(bit_shift, vsx_i16x8_const(
OJPH_REPEAT8(0x0101)));
1088 d0 = vsx_i16x8_mul(d0, bit_shift);
1089 d0 = vsx_u16x8_shr(d0, 8);
1090 d1 = vsx_i16x8_mul(d1, bit_shift);
1093 d0 = vsx_v128_or(d0, d1);
1099 v128_t U_q_m1 = vsx_i32x4_sub(U_q, ones);
1102 w0 = vsx_i16x8_sub(twos, w0);
1103 t0 = vsx_v128_and(w0, vsx_i64x2_const(-1, 0));
1104 t1 = vsx_v128_and(w0, vsx_i64x2_const(0, -1));
1105 t0 = vsx_i32x4_shl(t0, (
int)Uq0);
1106 t1 = vsx_i32x4_shl(t1, (
int)Uq1);
1107 shift = vsx_v128_or(t0, t1);
1108 ms_vec = vsx_v128_and(d0, vsx_i16x8_sub(shift, ones));
1111 w0 = vsx_v128_and(flags, vsx_i16x8_const(
OJPH_REPEAT8(0x800)));
1112 w0 = vsx_i16x8_eq(w0, vsx_i64x2_const(0, 0));
1113 w0 = vsx_v128_andnot(shift, w0);
1114 ms_vec = vsx_v128_or(ms_vec, w0);
1115 w0 = vsx_i16x8_shl(ms_vec, 15);
1116 ms_vec = vsx_v128_or(ms_vec, ones);
1118 ms_vec = vsx_i16x8_add(ms_vec, twos);
1119 ms_vec = vsx_i16x8_shl(ms_vec, (
int)p - 1);
1120 ms_vec = vsx_v128_or(ms_vec, w0);
1121 row = vsx_v128_andnot(ms_vec, insig);
1123 ms_vec = vsx_v128_andnot(tvn, insig);
1124 w0 = vsx_i8x16_swizzle(ms_vec,
1125 vsx_i16x8_const(0x0302, 0x0706, -1, -1, -1, -1, -1, -1));
1126 vn = vsx_v128_or(vn, w0);
1127 w0 = vsx_i8x16_swizzle(ms_vec,
1128 vsx_i16x8_const(-1, 0x0B0A, 0x0F0E, -1, -1, -1, -1, -1));
1129 vn = vsx_v128_or(vn, w0);
1131 pos += (
ui32)total_mn;
1155 ui32 missing_msbs,
ui32 num_passes,
1160 static bool insufficient_precision =
false;
1161 static bool modify_code =
false;
1162 static bool truncate_spp_mrp =
false;
1164 if (num_passes > 1 && lengths2 == 0)
1166 OJPH_WARN(0x00010001,
"A malformed codeblock that has more than "
1167 "one coding pass, but zero length for "
1168 "2nd and potential 3rd pass.\n");
1174 OJPH_WARN(0x00010002,
"We do not support more than 3 coding passes; "
1175 "This codeblocks has %d passes.\n",
1180 if (missing_msbs > 30)
1182 if (insufficient_precision ==
false)
1184 insufficient_precision =
true;
1185 OJPH_WARN(0x00010003,
"32 bits are not enough to decode this "
1186 "codeblock. This message will not be "
1187 "displayed again.\n");
1191 else if (missing_msbs == 30)
1193 if (modify_code ==
false) {
1195 OJPH_WARN(0x00010004,
"Not enough precision to decode the cleanup "
1196 "pass. The code can be modified to support "
1197 "this case. This message will not be "
1198 "displayed again.\n");
1202 else if (missing_msbs == 29)
1204 if (num_passes > 1) {
1206 if (truncate_spp_mrp ==
false) {
1207 truncate_spp_mrp =
true;
1208 OJPH_WARN(0x00010005,
"Not enough precision to decode the SgnProp "
1209 "nor MagRef passes; both will be skipped. "
1210 "This message will not be displayed "
1215 ui32 p = 30 - missing_msbs;
1221 OJPH_WARN(0x00010006,
"Wrong codeblock length.\n");
1227 lcup = (int)lengths1;
1229 scup = (((int)coded_data[lcup-1]) << 4) + (coded_data[lcup-2] & 0xF);
1230 if (scup < 2 || scup > lcup || scup > 4079)
1248 ui16 scratch[8 * 513] = {0};
1256 ui32 sstr = ((width + 2u) + 7u) & ~7u;
1258 assert((stride & 0x3) == 0);
1260 ui32 mmsbp2 = missing_msbs + 2;
1272 mel_init(&mel, coded_data, lcup, scup);
1274 rev_init(&vlc, coded_data, lcup, scup);
1284 for (
ui32 x = 0; x < width; sp += 4)
1303 t0 = (run == -1) ? t0 : 0;
1317 c_q = ((t0 & 0x10U) << 3) | ((t0 & 0xE0U) << 2);
1326 t1 =
vlc_tbl0[c_q + (vlc_val & 0x7F)];
1329 if (c_q == 0 && x < width)
1334 t1 = (run == -1) ? t1 : 0;
1339 t1 = x < width ? t1 : 0;
1348 c_q = ((t1 & 0x10U) << 3) | ((t1 & 0xE0U) << 2);
1356 ui32 uvlc_mode = ((t0 & 0x8U) << 3) | ((t1 & 0x8U) << 4);
1357 if (uvlc_mode == 0xc0)
1361 uvlc_mode += (run == -1) ? 0x40 : 0;
1378 ui32 len = uvlc_entry & 0xF;
1379 ui32 tmp = vlc_val & ((1 << len) - 1);
1383 len = uvlc_entry & 0x7;
1385 ui16 u_q = (
ui16)(1 + (uvlc_entry&7) + (tmp&~(0xFFU<<len)));
1387 u_q = (
ui16)(1 + (uvlc_entry >> 3) + (tmp >> len));
1393 for (
ui32 y = 2; y < height; y += 2)
1396 ui16 *sp = scratch + (y >> 1) * sstr;
1398 for (
ui32 x = 0; x < width; sp += 4)
1404 c_q |= ((sp[0 - (
si32)sstr] & 0xA0U) << 2);
1405 c_q |= ((sp[2 - (
si32)sstr] & 0x20U) << 4);
1421 t0 = (run == -1) ? t0 : 0;
1436 c_q = ((t0 & 0x40U) << 2) | ((t0 & 0x80U) << 1);
1438 c_q |= sp[0 - (
si32)sstr] & 0x80;
1440 c_q |= ((sp[2 - (
si32)sstr] & 0xA0U) << 2);
1441 c_q |= ((sp[4 - (
si32)sstr] & 0x20U) << 4);
1450 t1 =
vlc_tbl1[ c_q + (vlc_val & 0x7F)];
1453 if (c_q == 0 && x < width)
1458 t1 = (run == -1) ? t1 : 0;
1463 t1 = x < width ? t1 : 0;
1473 c_q = ((t1 & 0x40U) << 2) | ((t1 & 0x80U) << 1);
1475 c_q |= sp[2 - (
si32)sstr] & 0x80;
1483 ui32 uvlc_mode = ((t0 & 0x8U) << 3) | ((t1 & 0x8U) << 4);
1489 ui32 len = uvlc_entry & 0xF;
1490 ui32 tmp = vlc_val & ((1 << len) - 1);
1494 len = uvlc_entry & 0x7;
1496 ui16 u_q = (
ui16)((uvlc_entry & 7) + (tmp & ~(0xFFU << len)));
1498 u_q = (
ui16)((uvlc_entry >> 3) + (tmp >> len));
1521 const int v_n_size = 512 + 8;
1522 ui32 v_n_scratch[2 * v_n_size] = {0};
1529 ui32 *vp = v_n_scratch;
1530 ui32 *dp = decoded_data;
1533 for (
ui32 x = 0; x < width; x += 4, sp += 4, vp += 2, dp += 4)
1540 inf_u_q = vsx_v128_load(sp);
1541 U_q = vsx_u32x4_shr(inf_u_q, 16);
1543 w0 = vsx_i32x4_gt(U_q, vsx_u32x4_splat(mmsbp2));
1544 ui32 i = (
ui32)vsx_i8x16_bitmask(w0);
1552 w0 = vsx_v128_load(vp);
1553 w0 = vsx_v128_and(w0, vsx_i32x4_const(-1,0,0,0));
1554 w0 = vsx_v128_or(w0, vn);
1555 vsx_v128_store(vp, w0);
1563 vsx_v128_store(dp, row0);
1564 vsx_v128_store(dp + stride, row1);
1568 for (
ui32 y = 2; y < height; y += 2)
1572 ui32 *vp = v_n_scratch;
1573 const v128_t lut_lo = vsx_i8x16_const(
1574 31, 7, 6, 6, 5, 5, 5, 5, 4, 4, 4, 4, 4, 4, 4, 4
1576 const v128_t lut_hi = vsx_i8x16_const(
1577 31, 3, 2, 2, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0
1583 for (
ui32 x = 0; x <= width; x += 8, vp += 4)
1586 v = vsx_v128_load(vp);
1588 t = vsx_v128_and(nibble_mask, v);
1589 v = vsx_v128_and(vsx_u16x8_shr(v, 4), nibble_mask);
1590 t = vsx_i8x16_swizzle(lut_lo, t);
1591 v = vsx_i8x16_swizzle(lut_hi, v);
1592 v = vsx_u8x16_min(v, t);
1594 t = vsx_u16x8_shr(v, 8);
1595 v = vsx_v128_or(v, byte_offset8);
1596 v = vsx_u8x16_min(v, t);
1598 t = vsx_u32x4_shr(v, 16);
1599 v = vsx_v128_or(v, byte_offset16);
1600 v = vsx_u8x16_min(v, t);
1602 v = vsx_i16x8_sub(cc, v);
1603 vsx_v128_store(vp + v_n_size, v);
1607 ui32 *vp = v_n_scratch;
1608 ui16 *sp = scratch + (y >> 1) * sstr;
1609 ui32 *dp = decoded_data + y * stride;
1612 for (
ui32 x = 0; x < width; x += 4, sp += 4, vp += 2, dp += 4)
1619 v128_t gamma, emax, kappa, u_q;
1621 inf_u_q = vsx_v128_load(sp);
1623 vsx_v128_and(inf_u_q, vsx_i32x4_const(
OJPH_REPEAT4(0xF0)));
1624 w0 = vsx_i32x4_sub(gamma, vsx_i32x4_const(
OJPH_REPEAT4(1)));
1625 gamma = vsx_v128_and(gamma, w0);
1626 gamma = vsx_i32x4_eq(gamma, vsx_i64x2_const(0, 0));
1628 emax = vsx_v128_load(vp + v_n_size);
1630 emax = vsx_i16x8_max(w0, emax);
1631 emax = vsx_v128_andnot(emax, gamma);
1634 kappa = vsx_i16x8_max(emax, kappa);
1636 u_q = vsx_u32x4_shr(inf_u_q, 16);
1637 U_q = vsx_i32x4_add(u_q, kappa);
1639 w0 = vsx_i32x4_gt(U_q, vsx_u32x4_splat(mmsbp2));
1640 ui32 i = (
ui32)vsx_i8x16_bitmask(w0);
1648 w0 = vsx_v128_load(vp);
1649 w0 = vsx_v128_and(w0, vsx_i32x4_const(-1,0,0,0));
1650 w0 = vsx_v128_or(w0, vn);
1651 vsx_v128_store(vp, w0);
1658 vsx_v128_store(dp, row0);
1659 vsx_v128_store(dp + stride, row1);
1674 const int v_n_size = 512 + 8;
1675 ui16 v_n_scratch[2 * v_n_size] = {0};
1679 const ui32 dbuf_cap = 4096 * 15 / 8;
1680 ui8 dbuf[dbuf_cap + 72];
1687 ui16 *vp = v_n_scratch;
1688 ui32 *dp = decoded_data;
1691 for (
ui32 x = 0; x < width; x += 4, sp += 4, vp += 2, dp += 4)
1698 inf_u_q = vsx_v128_load(sp);
1699 U_q = vsx_u32x4_shr(inf_u_q, 16);
1701 w0 = vsx_i32x4_gt(U_q, vsx_u32x4_splat(mmsbp2));
1702 ui32 i = (
ui32)vsx_i8x16_bitmask(w0);
1709 w0 = vsx_v128_load(vp);
1710 w0 = vsx_v128_and(w0, vsx_i16x8_const(-1,0,0,0,0,0,0,0));
1711 w0 = vsx_v128_or(w0, vn);
1712 vsx_v128_store(vp, w0);
1715 w0 = vsx_i8x16_swizzle(row,
1716 vsx_i16x8_const(-1, 0x0100, -1, 0x0504,
1717 -1, 0x0908, -1, 0x0D0C));
1718 vsx_v128_store(dp, w0);
1719 w1 = vsx_i8x16_swizzle(row,
1720 vsx_i16x8_const(-1, 0x0302, -1, 0x0706,
1721 -1, 0x0B0A, -1, 0x0F0E));
1722 vsx_v128_store(dp + stride, w1);
1726 for (
ui32 y = 2; y < height; y += 2)
1730 ui16 *vp = v_n_scratch;
1731 const v128_t lut_lo = vsx_i8x16_const(
1732 15, 7, 6, 6, 5, 5, 5, 5, 4, 4, 4, 4, 4, 4, 4, 4
1734 const v128_t lut_hi = vsx_i8x16_const(
1735 15, 3, 2, 2, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0
1740 for (
ui32 x = 0; x <= width; x += 16, vp += 8)
1743 v = vsx_v128_load(vp);
1745 t = vsx_v128_and(nibble_mask, v);
1746 v = vsx_v128_and(vsx_u16x8_shr(v, 4), nibble_mask);
1747 t = vsx_i8x16_swizzle(lut_lo, t);
1748 v = vsx_i8x16_swizzle(lut_hi, v);
1749 v = vsx_u8x16_min(v, t);
1751 t = vsx_u16x8_shr(v, 8);
1752 v = vsx_v128_or(v, byte_offset8);
1753 v = vsx_u8x16_min(v, t);
1755 v = vsx_i16x8_sub(cc, v);
1756 vsx_v128_store(vp + v_n_size, v);
1760 ui16 *vp = v_n_scratch;
1761 ui16 *sp = scratch + (y >> 1) * sstr;
1762 ui32 *dp = decoded_data + y * stride;
1765 for (
ui32 x = 0; x < width; x += 4, sp += 4, vp += 2, dp += 4)
1772 v128_t gamma, emax, kappa, u_q;
1774 inf_u_q = vsx_v128_load(sp);
1776 vsx_v128_and(inf_u_q, vsx_i32x4_const(
OJPH_REPEAT4(0xF0)));
1777 w0 = vsx_i32x4_sub(gamma, vsx_i32x4_const(
OJPH_REPEAT4(1)));
1778 gamma = vsx_v128_and(gamma, w0);
1779 gamma = vsx_i32x4_eq(gamma, vsx_i64x2_const(0, 0));
1781 emax = vsx_v128_load(vp + v_n_size);
1783 vsx_i64x2_const(0, 0), 1, 2, 3, 4, 5, 6, 7, 8);
1784 emax = vsx_i16x8_max(w0, emax);
1785 emax = vsx_i8x16_swizzle(emax,
1786 vsx_i16x8_const(0x0100, -1, 0x0302, -1,
1787 0x0504, -1, 0x0706, -1));
1788 emax = vsx_v128_andnot(emax, gamma);
1791 kappa = vsx_i16x8_max(emax, kappa);
1793 u_q = vsx_u32x4_shr(inf_u_q, 16);
1794 U_q = vsx_i32x4_add(u_q, kappa);
1796 w0 = vsx_i32x4_gt(U_q, vsx_u32x4_splat(mmsbp2));
1797 ui32 i = (
ui32)vsx_i8x16_bitmask(w0);
1804 w0 = vsx_v128_load(vp);
1805 w0 = vsx_v128_and(w0, vsx_i16x8_const(-1,0,0,0,0,0,0,0));
1806 w0 = vsx_v128_or(w0, vn);
1807 vsx_v128_store(vp, w0);
1809 w0 = vsx_i8x16_swizzle(row,
1810 vsx_i16x8_const(-1, 0x0100, -1, 0x0504,
1811 -1, 0x0908, -1, 0x0D0C));
1812 vsx_v128_store(dp, w0);
1813 w1 = vsx_i8x16_swizzle(row,
1814 vsx_i16x8_const(-1, 0x0302, -1, 0x0706,
1815 -1, 0x0B0A, -1, 0x0F0E));
1816 vsx_v128_store(dp + stride, w1);
1830 ui16*
const sigma = scratch;
1832 ui32 mstr = (width + 3u) >> 2;
1834 mstr = ((mstr + 2u) + 7u) & ~7u;
1844 const v128_t shuffle_mask = vsx_i32x4_const(0x0C080400,-1,-1,-1);
1845 for (y = 0; y < height; y += 4)
1847 ui16* sp = scratch + (y >> 1) * sstr;
1848 ui16* dp = sigma + (y >> 2) * mstr;
1849 for (
ui32 x = 0; x < width; x += 8, sp += 8, dp += 2)
1851 v128_t s0, s1, u3, uC, t0, t1;
1853 s0 = vsx_v128_load(sp);
1854 u3 = vsx_v128_and(s0, mask_3);
1855 u3 = vsx_u32x4_shr(u3, 4);
1856 uC = vsx_v128_and(s0, mask_C);
1857 uC = vsx_u32x4_shr(uC, 2);
1858 t0 = vsx_v128_or(u3, uC);
1860 s1 = vsx_v128_load(sp + sstr);
1861 u3 = vsx_v128_and(s1, mask_3);
1862 u3 = vsx_u32x4_shr(u3, 2);
1863 uC = vsx_v128_and(s1, mask_C);
1864 t1 = vsx_v128_or(u3, uC);
1866 v128_t r = vsx_v128_or(t0, t1);
1867 r = vsx_i8x16_swizzle(r, shuffle_mask);
1875 ui16* dp = sigma + (y >> 2) * mstr;
1876 v128_t zero = vsx_i64x2_const(0, 0);
1877 for (
ui32 x = 0; x < width; x += 32, dp += 8)
1878 vsx_v128_store(dp, zero);
1894 ui16 prev_row_sig[256 + 8] = {0};
1897 frwd_init<0>(&sigprop, coded_data + lengths1, (
int)lengths2);
1899 for (
ui32 y = 0; y < height; y += 4)
1901 ui32 pattern = 0xFFFFu;
1902 if (height - y < 4) {
1904 if (height - y < 3) {
1914 ui16 *prev_sig = prev_row_sig;
1915 ui16 *cur_sig = sigma + (y >> 2) * mstr;
1916 ui32 *dpp = decoded_data + y * stride;
1917 for (
ui32 x = 0; x < width; x += 4, dpp += 4, ++cur_sig, ++prev_sig)
1922 pattern = pattern >> (s * 4);
1936 ui32 ps; memcpy(&ps, prev_sig,
sizeof(ps));
1937 ui32 ns; memcpy(&ns, cur_sig + mstr,
sizeof(ns));
1938 ui32 u = (ps & 0x88888888) >> 3;
1940 u |= (ns & 0x11111111) << 3;
1942 ui32 cs; memcpy(&cs, cur_sig,
sizeof(cs));
1945 mbr |= (cs & 0x77777777) << 1;
1946 mbr |= (cs & 0xEEEEEEEE) >> 1;
1966 ui32 col_mask = 0xFu;
1967 ui32 inv_sig = ~cs & pattern;
1968 for (
int i = 0; i < 16; i += 4, col_mask <<= 4)
1970 if ((col_mask & new_sig) == 0)
1974 ui32 sample_mask = 0x1111u & col_mask;
1975 if (new_sig & sample_mask)
1977 new_sig &= ~sample_mask;
1980 ui32 t = 0x33u << i;
1981 new_sig |= t & inv_sig;
1987 if (new_sig & sample_mask)
1989 new_sig &= ~sample_mask;
1992 ui32 t = 0x76u << i;
1993 new_sig |= t & inv_sig;
1999 if (new_sig & sample_mask)
2001 new_sig &= ~sample_mask;
2004 ui32 t = 0xECu << i;
2005 new_sig |= t & inv_sig;
2011 if (new_sig & sample_mask)
2013 new_sig &= ~sample_mask;
2016 ui32 t = 0xC8u << i;
2017 new_sig |= t & inv_sig;
2027 v128_t new_sig_vec = vsx_i16x8_splat((
si16)new_sig);
2028 new_sig_vec = vsx_i8x16_swizzle(new_sig_vec,
2029 vsx_i8x16_const(0,0,0,0,0,0,0,0,1,1,1,1,1,1,1,1));
2030 new_sig_vec = vsx_v128_and(new_sig_vec,
2032 new_sig_vec = vsx_i8x16_eq(new_sig_vec,
2037 v128_t ex_sum, shfl, inc_sum = new_sig_vec;
2038 inc_sum = vsx_i8x16_abs(inc_sum);
2040 15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30);
2041 inc_sum = vsx_i8x16_add(inc_sum, shfl);
2043 7, 8, 9, 10, 11, 12, 13, 14);
2044 inc_sum = vsx_i8x16_add(inc_sum, shfl);
2047 inc_sum = vsx_i8x16_add(inc_sum, shfl);
2050 inc_sum = vsx_i8x16_add(inc_sum, shfl);
2054 15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30);
2058 cwd_vec = vsx_i16x8_splat((
si16)cwd);
2059 cwd_vec = vsx_i8x16_swizzle(cwd_vec,
2060 vsx_i8x16_const(0,0,0,0,0,0,0,0,1,1,1,1,1,1,1,1));
2061 cwd_vec = vsx_v128_and(cwd_vec,
2063 cwd_vec = vsx_i8x16_eq(cwd_vec,
2065 cwd_vec = vsx_i8x16_abs(cwd_vec);
2069 v128_t v = vsx_i8x16_swizzle(cwd_vec, ex_sum);
2072 v128_t m = vsx_i8x16_const(
2073 0,-1,-1,-1,4,-1,-1,-1,8,-1,-1,-1,12,-1,-1,-1);
2074 v128_t val = vsx_i32x4_splat(3 << (p - 2));
2076 for (
int c = 0; c < 4; ++ c) {
2077 v128_t s0, s0_ns, s0_val;
2079 s0 = vsx_v128_load(dp);
2083 s0_ns = vsx_i8x16_swizzle(new_sig_vec, m);
2084 s0_ns = vsx_i32x4_eq(s0_ns,
2088 s0_val = vsx_i8x16_swizzle(v, m);
2089 s0_val = vsx_i32x4_shl(s0_val, 31);
2090 s0_val = vsx_v128_or(s0_val, val);
2091 s0_val = vsx_v128_and(s0_val, s0_ns);
2094 s0 = vsx_v128_or(s0, s0_val);
2096 vsx_v128_store(dp, s0);
2099 m = vsx_i32x4_add(m, vsx_i32x4_const(
OJPH_REPEAT4(1)));
2106 *prev_sig = (
ui16)(new_sig);
2110 new_sig |= (t & 0x7777) << 1;
2111 new_sig |= (t & 0xEEEE) >> 1;
2124 rev_init_mrp(&magref, coded_data, (
int)lengths1, (
int)lengths2);
2126 for (
ui32 y = 0; y < height; y += 4)
2128 ui16 *cur_sig = sigma + (y >> 2) * mstr;
2129 ui32 *dpp = decoded_data + y * stride;
2130 for (
ui32 i = 0; i < width; i += 4, dpp += 4)
2135 ui16 sig = *cur_sig++;
2144 sig_vec = vsx_i8x16_swizzle(sig_vec,
2145 vsx_i8x16_const(0,0,0,0,0,0,0,0,1,1,1,1,1,1,1,1));
2146 sig_vec = vsx_v128_and(sig_vec,
2148 sig_vec = vsx_i8x16_eq(sig_vec,
2150 sig_vec = vsx_i8x16_abs(sig_vec);
2154 v128_t ex_sum, shfl, inc_sum = sig_vec;
2156 15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30);
2157 inc_sum = vsx_i8x16_add(inc_sum, shfl);
2159 7, 8, 9, 10, 11, 12, 13, 14);
2160 inc_sum = vsx_i8x16_add(inc_sum, shfl);
2163 inc_sum = vsx_i8x16_add(inc_sum, shfl);
2166 inc_sum = vsx_i8x16_add(inc_sum, shfl);
2170 15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30);
2178 cwd_vec = vsx_i8x16_swizzle(cwd_vec,
2179 vsx_i8x16_const(0,0,0,0,0,0,0,0,1,1,1,1,1,1,1,1));
2180 cwd_vec = vsx_v128_and(cwd_vec,
2182 cwd_vec = vsx_i8x16_eq(cwd_vec,
2186 cwd_vec = vsx_i8x16_add(cwd_vec, cwd_vec);
2191 v128_t m = vsx_i8x16_const(0,-1,-1,-1,4,-1,-1,-1,
2192 8,-1,-1,-1,12,-1,-1,-1);
2194 for (
int c = 0; c < 4; ++c) {
2195 v128_t s0, s0_sig, s0_idx, s0_val;
2197 s0 = vsx_v128_load(dp);
2199 s0_sig = vsx_i8x16_swizzle(sig_vec, m);
2200 s0_sig = vsx_i8x16_eq(s0_sig, vsx_i64x2_const(0, 0));
2202 s0_idx = vsx_i8x16_swizzle(ex_sum, m);
2203 s0_val = vsx_i8x16_swizzle(cwd_vec, s0_idx);
2205 s0_val = vsx_v128_andnot(s0_val, s0_sig);
2207 s0_val = vsx_i32x4_shl(s0_val, (
int)p - 2);
2208 s0 = vsx_v128_xor(s0, s0_val);
2210 vsx_v128_store(dp, s0);
2213 m = vsx_i32x4_add(m, vsx_i32x4_const(
OJPH_REPEAT4(1)));