2
0

idct4x4_msa.c 2.9 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798
  1. /*
  2. * Copyright (c) 2015 The WebM project authors. All Rights Reserved.
  3. *
  4. * Use of this source code is governed by a BSD-style license
  5. * that can be found in the LICENSE file in the root of the source
  6. * tree. An additional intellectual property rights grant can be found
  7. * in the file PATENTS. All contributing project authors may
  8. * be found in the AUTHORS file in the root of the source tree.
  9. */
  10. #include "vpx_dsp/mips/inv_txfm_msa.h"
  11. void vpx_iwht4x4_16_add_msa(const int16_t *input, uint8_t *dst,
  12. int32_t dst_stride) {
  13. v8i16 in0, in1, in2, in3;
  14. v4i32 in0_r, in1_r, in2_r, in3_r, in4_r;
  15. /* load vector elements of 4x4 block */
  16. LD4x4_SH(input, in0, in2, in3, in1);
  17. TRANSPOSE4x4_SH_SH(in0, in2, in3, in1, in0, in2, in3, in1);
  18. UNPCK_R_SH_SW(in0, in0_r);
  19. UNPCK_R_SH_SW(in2, in2_r);
  20. UNPCK_R_SH_SW(in3, in3_r);
  21. UNPCK_R_SH_SW(in1, in1_r);
  22. SRA_4V(in0_r, in1_r, in2_r, in3_r, UNIT_QUANT_SHIFT);
  23. in0_r += in2_r;
  24. in3_r -= in1_r;
  25. in4_r = (in0_r - in3_r) >> 1;
  26. in1_r = in4_r - in1_r;
  27. in2_r = in4_r - in2_r;
  28. in0_r -= in1_r;
  29. in3_r += in2_r;
  30. TRANSPOSE4x4_SW_SW(in0_r, in1_r, in2_r, in3_r, in0_r, in1_r, in2_r, in3_r);
  31. in0_r += in1_r;
  32. in2_r -= in3_r;
  33. in4_r = (in0_r - in2_r) >> 1;
  34. in3_r = in4_r - in3_r;
  35. in1_r = in4_r - in1_r;
  36. in0_r -= in3_r;
  37. in2_r += in1_r;
  38. PCKEV_H4_SH(in0_r, in0_r, in1_r, in1_r, in2_r, in2_r, in3_r, in3_r, in0, in1,
  39. in2, in3);
  40. ADDBLK_ST4x4_UB(in0, in3, in1, in2, dst, dst_stride);
  41. }
  42. void vpx_iwht4x4_1_add_msa(const int16_t *input, uint8_t *dst,
  43. int32_t dst_stride) {
  44. int16_t a1, e1;
  45. v8i16 in1, in0 = { 0 };
  46. a1 = input[0] >> UNIT_QUANT_SHIFT;
  47. e1 = a1 >> 1;
  48. a1 -= e1;
  49. in0 = __msa_insert_h(in0, 0, a1);
  50. in0 = __msa_insert_h(in0, 1, e1);
  51. in0 = __msa_insert_h(in0, 2, e1);
  52. in0 = __msa_insert_h(in0, 3, e1);
  53. in1 = in0 >> 1;
  54. in0 -= in1;
  55. ADDBLK_ST4x4_UB(in0, in1, in1, in1, dst, dst_stride);
  56. }
  57. void vpx_idct4x4_16_add_msa(const int16_t *input, uint8_t *dst,
  58. int32_t dst_stride) {
  59. v8i16 in0, in1, in2, in3;
  60. /* load vector elements of 4x4 block */
  61. LD4x4_SH(input, in0, in1, in2, in3);
  62. /* rows */
  63. TRANSPOSE4x4_SH_SH(in0, in1, in2, in3, in0, in1, in2, in3);
  64. VP9_IDCT4x4(in0, in1, in2, in3, in0, in1, in2, in3);
  65. /* columns */
  66. TRANSPOSE4x4_SH_SH(in0, in1, in2, in3, in0, in1, in2, in3);
  67. VP9_IDCT4x4(in0, in1, in2, in3, in0, in1, in2, in3);
  68. /* rounding (add 2^3, divide by 2^4) */
  69. SRARI_H4_SH(in0, in1, in2, in3, 4);
  70. ADDBLK_ST4x4_UB(in0, in1, in2, in3, dst, dst_stride);
  71. }
  72. void vpx_idct4x4_1_add_msa(const int16_t *input, uint8_t *dst,
  73. int32_t dst_stride) {
  74. int16_t out;
  75. v8i16 vec;
  76. out = ROUND_POWER_OF_TWO((input[0] * cospi_16_64), DCT_CONST_BITS);
  77. out = ROUND_POWER_OF_TWO((out * cospi_16_64), DCT_CONST_BITS);
  78. out = ROUND_POWER_OF_TWO(out, 4);
  79. vec = __msa_fill_h(out);
  80. ADDBLK_ST4x4_UB(vec, vec, vec, vec, dst, dst_stride);
  81. }