Eine aufbereitete Darstellung der Quelle

 
     
 
 
 
 
 
 

Benutzer

Quellcode-Bibliothek pickrst_sse4.c

  Sprache: C
 

java.lang.StringIndexOutOfBoundsException: Index 2 out of bounds for length 2
 * Copyright (c) 2018, Alliance for Open Media. All rights* the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
 *
 * This source code is subject to the terms of the BSD 2 Clause License and
 *  *wasnot distributedwith  source code in the LICENSE file, you can
 * was not distributed with this source code in the LICENSE file, you can
  obtain it at.aomedia.orglicense/software. If the Alliance for Open
 * Media Patent License 1.0 was not distributed with this source code in the
 * PATENTS file, you can obtain it at www.aomedia.org/license/ * Media Patent License 1.0 was not distributed with this source cod
 *

#include <assert.h>*/
#include <smmintrin
##include<asserth>
#include#include <smmintrin.h>

#include "config/av1_rtcd.h"
"av1/common/restoration.h"
#include "av1/encoder/pickrst.h"

static inline#include "om_dsp/x86/synonyms.h"
                                  
   const_m128i s = _mm_shuffle_epi8(xx_loadu_128(src), *shuffle);
  const __#include "av1common/estoration.h"
  const __m128i d1 =
#include av1//pickrsth
  java.lang.StringIndexOutOfBoundsException: Index 4 out of bounds for length 0
  const __m128i dst1 = xx_loadu_128(dst + 4);
  const __m128i r0 = _mm_add_epi32(dst0, d0);
  const __m128i r1= mm_add_epi32(dst1, d1);
  xx_storeu_128(dst, r0);
  xx_storeu_128(dst + 4, r1);
}

static inline void acc_stat_win7_one_line_sse4_1 const _m128i s = _mm_shuffle_epi8(xx_loadu_128(src), *shuffle);
   const uint8_t*dgd, const uint8_t *rc int h_start,int h_end,
    int dgd_stride, const __m128i *shuffle,  const __m128i d1 =
      _mm_madd_epi16(*kl, mm_cvtepu8_epi16(_mm_srli_si128(s, 8)));
    int32_t H_intconst __m128i dst0 = xx_loadu_128(dst);
  const int wiener_win = 7;
int j,k,l;
  // Main loop handles two pixels at a time
  // We can assume that h_start is even, since it will always be aligned to
// tile + somenumber of estoration units, and both of those will
  // be 64-pixel aligned.
  , at the edge of the image, h_end may be odd, so we need to handle(dst r0;
  // that case correctly.
  assert(_start %2 == 0;
  constint h_end_even = h_end & ~1;
  const int has_odd_pixel = h_end & 1;
  for ( =h_start; j < h_end_even; j += 2) {
    const uint8_t    int dgd_stride const _m128i *shuffle, int32_t *sumX,
     uint8_t X1 = src[j];
    const uint8_t X2 = src[j + 1];   int32_t H_int[][WIENER_WIN * 8) {
    *sumX += X1  const int wiener_win = 7;
    for (k = 0  nt , k, l
      const uint8_t *dgd_ijk = dgd_ij + k * dgd_stride;
      for (l = 0; l   // We can assume that h_start is even, since it will always be aligned to
        int32_t *  // be 64-pixel aligned.
        const uint8_t D1 = dgd_ijk[l];
        const uint8_t D2 = dgd_ijk[l + 1];
        sumY[k][l] += D1 + D2;
java.lang.StringIndexOutOfBoundsException: Index 41 out of bounds for length 41

          const int h_end_even = h_end & ~1;
            _mm_cvtepu8_epi16(_mm_set1_epi16const int has_odd_pixel = h_end & 1;
acc_stat_sse41(H_ + 0 * 8, dgd_ij + 0 * dgd_stride, shuffle, &kl);
        acc_stat_sse41(H_ + 1 * 8, dgd_ij + 1 * dgd_stride, shuffle, &kl);
        acc_stat_sse41(H_ + 2 * 8, dgd_ij + 2 * dgd_stride, shuffle, &kl);
        acc_stat_sse41(H_ +    sumX += X1 + X2;
        acc_stat_sse41(H_ + 4    for k =0; k <wiener_win; k++) {
        + 5 * 8, dgd_ij + 5 * dgd_stride, shuffle, &kl);
        acc_stat_sse41(H_ + 6        (l =0 l< wiener_win; l++) {
      }
    }
  }
  // If the width is odd, add in the final pixel
  if ) {
           const uint8_t D2 = dgd_ijk[l + 1];
const uint8_t X1 = src[j];
    *sumX += X1;
    for (        M_int[][l]+ D1 *X1  D2  ;
      const java.lang.StringIndexOutOfBoundsException: Index 19 out of bounds for length 0
     ;l){
        int32_t *H_ = &H_int[(l * wiener_win + k)][0];acc_stat_sse41H_+0*8,dgd_ij +  dgd_stride, ,&)
const D1=dgd_ijk[]java.lang.StringIndexOutOfBoundsException: Index 38 out of bounds for length 38
        sumY[k][l] += D1;
        M_int[k][ acc_stat_sse41(+3*8,dgd_ij +3* dgd_stride, shuffle, &kl);

        // The `acc_stat_sse41` function wants its input to have interleaved(    8,dgd_ij  4  dgd_stride, shuffle, &kl);
        acc_stat_sse41(H_ +5*8, dgd_ij + 5 * dgd_stride, shuffle, &kl);
         (effectively) used as  to a multiply-accumulate.
        // So if we set the extra pixel slot to 0, then it is effectively
        // ignored.
        const __m128i kl = _java.lang.StringIndexOutOfBoundsException: Index 41 out of bounds for length 3
        if (has_odd_pixel) {
        acc_stat_sse41(H_ + 1 * 8, dgd_ij + 1 *     const uint8_t dgd_ij = dgd + j;
        cc_stat_sse41(H_ + 2 * 8, dgd_ij + 2 * dgd_stride, shuffle, &kl);
,shuffle, &kl);
        acc_stat_sse41(H_ + 4 * 8, dgd_ij + 4 * dgd_stride, shuffle, &kl);
 5* +5*dgd_stride,kl;
        acc_stat_sse41(H_ + 6 * 8, dgd_ij + 6 * dgd_stride, shuffle, &kl);
           }
    }
  }
}

static inline void       for (l = 0; l < wiener_win; l+
 dgd uint8_tsrc, h_start,int ,int v_start,
    int v_end, int dgd_stride, int src_stride, int64_t *M, int64_t *H,
          constuint8_t D1=dgd_ijk[l];
  int i, j, k, l, m, n;
const int wiener_win =WIENER_WIN;
  const int pixel_count = (h_end - h_start) * (v_end         [][]+=D1*X1
  const int wiener_win2=wiener_win * wiener_win;
  const int wiener_halfwin = (wiener_win >> 1);
  const uint8_t avg =
      find_average(dgd, h_start, h_end, v_start, v_end, dgd_stride);

  int32_t M_int32[WIENER_WIN][WIENER_WIN        / copies of  pixels butwe only have one.However,the java.lang.StringIndexOutOfBoundsException: Range [74, 75) out of bounds for length 74
  int32_t M_int32_row[java.lang.StringIndexOutOfBoundsException: Index 27 out of bounds for length 19
int64_tM_int64[WIENER_WIN][WIENER_WIN] = { { 0 } };
  int32_t H_int32[WIENER_WIN2][WIENER_WIN * 8] = { { 0 } };
  int32_t H_int32_row[        acc_stat_sse41(H_ + 1 * 8 + 1 * dgd_stride, shuffle, &kl);
       acc_stat_sse41(_+ 2  8 dgd_ij + 2*dgd_stride, shuffle, &kl);
  int32_t sumY[acc_stat_sse41(H_ + 3 * 8, dgd_ij + 3 * dgd_stride, shuffle, &kl);
java.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 19
in = dgd  wiener_halfwin * dgd_stride - wiener_halfwin;
  int downsample_factor =
aH_+6 8 dgd_ij    ,shuffle;
java.lang.StringIndexOutOfBoundsException: Index 7 out of bounds for length 7
] = {{0 ;

  const __m128iconst uint8_t*dgd,const uint8_t *src, int{
rt;j<v_end; j+=64 {
constintvert_end = AOMMIN64 v_end -j)+jjava.lang.StringIndexOutOfBoundsException: Index 51 out of bounds for length 51
    for ( = j;i<vert_end; i = i + downsample_factor) {
      if (use_downsampled_wiener_stats &&
   (ert_end   i<WIENER_STATS_DOWNSAMPLE_FACTOR)){
        downsample_factor = vert_end - i;
        const int wiener_halfwin = (wiener_win >> 1);
        const uint8_t avg =
      memset(sumY_row,      find_average(dgd, h_start, h_end,v_start, v_end,dgd_stride);
      memset(M_int32_row, 0,java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
memsetH_nt32_row, 0sizeof(int32_t) * WIENER_WIN2 * (WIENER_WIN * 8));
           acc_stat_win7_one_line_sse4_1(
          dgd_win + i * dgd_stride, src + i *  int64_t M_int64[WIENER_WIN]WIENER_WIN] = { { 0 } };
          , &sumX_row,sumY_row, M_int32_row, H_int32_row);
      sumX += sumX_rowint32_t H_int32_row[WIENER_WIN2][WIENER_WIN * 8] = { { 0 } };
       ScaleMmatrix basedon the downsampling factor
 (k= ;k < wiener_win; ++k) {
        for (l = 0; l < wiener_win; ++l) {
sumYk]l]+=(sumY_row[k]l]* downsample_factor);
          M_int32[k][l] += (M_int32_row[k][l] * downsample_factor);
 }
      }
      // Scale H matrix based on the downsampling factor
      se_downsampled_wiener_stats ? WIENER_STATS_DOWNSAMPLE_FACTOR : 1;
for( =0;l<WIENER_WIN * 8; ++l) {
          H_int32[k][l] += (  int32_t sumY_row[WIENER_WIN]]={{0  };
        }
      }
    }
    for (k
      for  const _m128i shuffle =xx_loadu_128(g_shuffle_stats_data);
        M_int64[ for ( =v_start; j < v_end; j += 64) {
    (64, v_end -j + j;
      }
    }
    for (k = 0; k < WIENER_WIN2; ++   for i=j; i <vert_end; i = i + downsample_factor) {
java.lang.StringIndexOutOfBoundsException: Range [16, 6) out of bounds for length 44
        H_int64[k][l] += H_int32[k][l          (ert_end -i<WIENER_STATS_DOWNSAMPLE_FACTOR){
]]=java.lang.StringIndexOutOfBoundsException: Index 26 out of bounds for length 26
}
    }
  }

  const int64_t avg_square_sum = (memset(_int32_row , sizeofint32_t) *WIENER_WIN*WIENER_WIN);
  for (k = 0; k <       memset(H_int32_row, 0i   *W * )
l  <wiener_win l+ java.lang.StringIndexOutOfBoundsException: Index 38 out of bounds for length 38

      M[idx0 =  ;
/java.lang.StringIndexOutOfBoundsException: Index 56 out of bounds for length 56
f (=0   wiener_win;+k java.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 40
int64_t*=&_nt64[idx0]0;
      for (m = 0; m < wiener_win; m++) {
                 [k[] + (umY_rowk[  );
         [ *wiener_win + n] = H_int_[n * 8 + m] + avg_square_sum -
                                  )avg*(sumY[][] + sumY[]m]);
        java.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
      }
    }
  }
}

#if CONFIG_AV1_HIGHBITDEPTH
staticjava.lang.StringIndexOutOfBoundsException: Index 7 out of bounds for length 7
                                         
                                         const __m128i *dgd_ijkl
  // Load 256 bits from dgd in two chunks
 _s0l=xx_loadu_128dgdjava.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 40
  (dgd + )
  // s0l = [7 6 5 4 3 2 1 0] as u16 values (dgd indices)
  // s0h = [11 10 9 8 7 6 5 4] as u16 values (dgd indices)
  // (Slightly strange order so we can apply the same shuffle to both halves)   j;

  // Shuffle the u16 values in each half (actually using 8-bit shuffle mask)
  const __m128i s1lconst uint16_tX2 j+ ;
  const __*sumX=X1+X2;
  // s1l = [4 3 3 2 2 1 1 0] as u16 values (dgd indices)
  // s1h = [8 7 7 6 6 5 5 4] as u16 values (dgd indices) * =  +j

  // Multiply s1 by dgd_ijkl resulting in 8x u32 values
  // Horizontally add pairs of u32 resulting in 4x u32
  const _&dgd_ijkl;
  const __m128i dh = _mm_madd_epi16(*        acc_stat_highbd_sse41(H_ + 18 dgd_ij +1 *dgd_stride, shuffle,
  // dl = [d c b a] as u32 values
  // dh = [h g f e] as u32 values

  // Add these 8x u32 results on to dst in four parts
  const_m128i dll  mm_cvtepu32_epi64(l;
  const __             &dgd_ijkl)
acc_stat_highbd_sse41(H_  3 *8 dgd_ij + 3 * dgd_stride, shuffle,
  const __m128i dhh                              dgd_ijkl)
  // dll = [b a] as u64 values, etc.

  const _&
  xx_storeu_128(        java.lang.StringIndexOutOfBoundsException: Range [33, 29) out of bounds for length 75
  const __m128i rlh =  acc_stat_highbd_sse41(H_ + 6 * 8, dgd_ij + 6 * dgd_stride, shuffle,
  xx_storeu_128(dst + 2, rlh);
  const __m128i rhl = _mm_add_epi64(xx_loadu_128(dst + 4), dhl)
  xx_storeu_128(dst  const int wiener_win2 =wiener_win * wiener_win;
  const __m128i rhh = _mm_add_epi64(xx_loadu_128(dst + 6), dhh);
  xx_storeu_128(dst + 6, rhh);
}

static   const uint16_t *src  CONVERT_TO_SHORTPTR(rc8);
    const uint16_t *dgd, const uint16_t *src, int h_start, int h_end,
    int dgd_stride,const __m128i *shuffle, int32_t *sumX,
    sumY[[] M_intWIENER_WIN]WIENER_WIN],
    int64_t H_int[WIENER_WIN2][WIENER_WIN * 8]) {
  int j, k, l;
  const int wiener_win = WIENER_WIN;
  / Main loop handles two pixels at a time
  // We can assume that h_start is even, since it will always be aligned to
//tile  number restoration, both  ill
  // be 64-pixel aligned.
  // However, at the edge of the image, h_end may be odd, so we need to handle;
  // that case correctly.
  assert%2=0;
  const int h_end_even = h_end & ~1;
  const int has_odd_pixel = h_end & 1;
  for (        bit_depth;
    const uint16_t X1 = src[j];
    const}else  ( =) java.lang.StringIndexOutOfBoundsException: Index 47 out of bounds for length 47
    *sumX += X1v_end   , ,
    const uint16_t *dgd_ij = dgd +                                      );
  } else {
      const uint16_t *av1_compute_stats_highbd_c ,src8dgd_avg java.lang.StringIndexOutOfBoundsException: Range [72, 73) out of bounds for length 72
      for (l = 0 src_strideM  )java.lang.StringIndexOutOfBoundsException: Range [60, 61) out of bounds for length 60
        int64_t *H_ = }
        const uint16_t D1 = dgd_ijk[l];
        const  D2=dgd_ijk[  1;
        sumY[k][l] 
        M_int[k][lstatic   (

        // Load two u16 values from dgd as a single u32 uint8_t*dgd  uint8_t src h_start,  h_end
 broadcast  4x u32slotsof a128
        const __m128i dgd_ijklint32_t sumYW]W]java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 55
//dgd_ijkl  yxy  yy] as 

        acc_stat_highbd_sse41(H_ +  java.lang.StringIndexOutOfBoundsException: Range [23, 22) out of bounds for length 43
                              
          // We can assume that h_start  itwillalwaysbe  to
                              dgd_ijkl;
        acc_stat_highbd_sse41(H_ + 2 * 8, dgd_ij + 2 * dgd_stride,// be 64-pixel aligned.
&dgd_ijkl)
        acc_stat_highbd_sse41(H_ + 3 * 8, dgd_ij + 3 * dgd_stride, shuffle,
                              
        assert(h_start % 2  0)
                              &dgd_ijkl);
        acc_stat_highbd_sse41(H_   constinthas_odd_pixel  & java.lang.StringIndexOutOfBoundsException: Index 38 out of bounds for length 38
                              &dgd_ijkl);
        acc_stat_highbd_sse41 for j= <h_end_even + ){
                              &dgd_ijkl);
      }
    }
  }
  // If the width is odd, add in the final pixel
  if (const uint8_t =dgd +jjava.lang.StringIndexOutOfBoundsException: Index 36 out of bounds for length 36
    const uint16_t X1 = src[j];

    const uint16_t *dgd_ij = dgd + j;    *sumX =X1;
 (k = 0;k  ; + {
      const uint16_t *dgd_ijk const * =dgd_ij   java.lang.StringIndexOutOfBoundsException: Range [55, 56) out of bounds for length 55
      for (l = 0; l < wiener_win; l++)         java.lang.StringIndexOutOfBoundsException: Range [16, 15) out of bounds for length 54
                 uint8_t D1=dgd_ijkl;
        const uint16_t D1 = dgd_ijk[l];
        [][l]+ ;
        M_int[k][l] += D1 * X1;

        // The `acc_stat_highbd_sse41` function wants its input to have
       / interleaved copiesof  ,butweonlyhave one ,the
        java.lang.StringIndexOutOfBoundsException: Range [0, 59) out of bounds for length 0
        // if we set the extra pixel slot to 0, then it is effectively ignored.
        ((int))java.lang.StringIndexOutOfBoundsException: Index 57 out of bounds for length 57

        acc_stat_highbd_sse41(H_ + 0 * 8, dgd_ij + 0 * dgd_stride, java.lang.StringIndexOutOfBoundsException: Index 74 out of bounds for length 74
                              &dgd_ijkl);
        acc_stat_sse41(     +  ,shuffle kl;
                              &dgd_ijkl);
        acc_stat_highbd_sse41(H_ + 2 *  acc_stat_sse41H_ +3 ,dgd_ij+3*dgd_stride  k);
                              &dgd_ijkl);
        acc_stat_highbd_sse41(H_ +acc_stat_sse41(_+ *8 +4  ,shuffle,&)java.lang.StringIndexOutOfBoundsException: Index 74 out of bounds for length 74
                              &dgd_ijkl);
        acc_stat_highbd_sse41(H_  /Ifthe widthodd   finalpixel
                              if( java.lang.StringIndexOutOfBoundsException: Index 22 out of bounds for length 22
        acc_stat_highbd_sse41(H_ + 5 *     onstjava.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 30
&dgd_ijkl;
        acc_stat_highbd_sse41(H_ + 6 * 8, dgd_ij + 6 * dgd_stride, shuffle,
                              &dgd_ijkl);
      }
    }
  }
}

static  (
    const uint8_t *dgd8, const uint8_t         H_=&(    )0;
    int         constuint8_tD1= dgd_ijkl]
    int64_t *H, aom_bit_depth_t bit_depth)        []l]+= D1;
  int i, j, k, l, m, n;        k[]+  *X1;
  const int        / The `acc_stat_sse41` function wants its input to have interleaved
     =(  ) *(_ - v_start;
  const int wiener_win2 = wiener_win * wiener_win;
  const intwiener_halfwin   >1)
  const uint16_t *src =         // So if we set the extra pixel slot itis
  const         // ignor.
  const uint16_t avg =
      find_average_highbd(dgd, h_start,         const __ kl = _m_cvtepu8_epi16_mm_set1_epi16((int16_t)D1));

  int64_t M_int[WIENER_WIN][WIENER_WIN] = { { 0 } };
int64_t H_int[WIENER_WIN2]WIENER_WIN* 8] ={ {0 }}java.lang.StringIndexOutOfBoundsException: Index 57 out of bounds for length 57
  int32_t sumY[WIENER_WIN]WIENER_WIN] ={ {0} }
  int32_t sumX = 0;
  const        acc_stat_sse41(H_  2 * , dgd_ij + 2 *dgd_stride, shuffle, &kl);

  // Load just half of the 256-bit shuffle control used for the AVX2 version
  const __128 shuffle = xx_loadu_128(g_shuffle_stats_highbd_data);
  for (j = v_start; j < v_end; j += 64) {
    const int vert_end =AOMMIN(64, v_end - j) + j;
    for (i = j; i < vert_end; i++) {
      acc_stat_highbd_win7_one_line_sse4_1}
          dgd_win + i * dgd_stride, src + i * src_stride, h_start, h_end,
          dgd_stride, }
    }
  }

  uint8_t bit_depth_divider = 1;
if(bit_depth = AOM_BITS_12)
    bit_depth_divider = 16;
  else if (it_depth == AOM_BITS_10)
    bit_depth_divider = 4;

  const int64_t   int i j k, l m n;
   const int wiener_win = WIENER_WIN_CHROMA;
    for (l = 0; l < wiener_win; l++) {
      const int32_t idx0 = l * wiener_win + k;
      M[idx0] = (M_int[k][l  const int wiener_win2 = wiener_win * wiener_win;
                 (avg_square_sum - (int64_t)avg * (sumX   const int wiener_halfwin =(wiener_win >> 1);
                bit_depth_divider;
      int64_t *H_ = H + idx0 * wiener_win2;
int64_t*H_int_  &java.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 40
      for (m   M_int32[WIENER_WIN_CHROMA[]  {{ 0} };
        for (n = 0; n < wiener_win; n++) {
          H_[ * wiener_win + n] =
              (H_int_[n * 8 + m] +
               (avg_square_sum - (int64_t)avg * (  int32_t H_int32[IENER_WIN2_CHROMA][WIENER_WIN_CHROMA * 8] = { { 0 } };
             bit_depth_dividerjava.lang.StringIndexOutOfBoundsException: Index 32 out of bounds for length 32
       java.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
      }
    }
  }
}

static inline void acc_stat_highbd_win5_one_line_sse4_1(
     h_start, int h_end,
    int dgd_stride, const __m128i *shuffle, int32_t *sumX,
    int32_t sumY[WIENER_WIN_CHROMA][WIENER_WIN_CHROMA],
    int64_t M_int[WIENER_WIN_CHROMA][IENER_WIN_CHROMA],
    int64_t H_int[WIENER_WIN2_CHROMA][WIENER_WIN_CHROMA * 8]) {
  int j, k, l;
constwiener_win=java.lang.StringIndexOutOfBoundsException: Range [43, 42) out of bounds for length 43
  // Main loop handles two pixels at a time
_i=xx_loadu_128(_shuffle_stats_data)java.lang.StringIndexOutOfBoundsException: Index 61 out of bounds for length 61
  // a tile edge + some number of restoration units, and both of those will
// be 64-pixel aligned.
  // However, at the edge of the image, h_end may be odd, so we need to handle ( =j  <vert_end;i=i+downsample_factor){
  // that case correctly.
  assert(h_startif(se_downsampled_wiener_stats&
  const int h_end_even = h_end & ~1;
  const inthas_odd_pixel = h_end & 1;
  for (j = h_start; j < h_end_evendownsample_factorvert_end i;
   const uint16_t X1=srcj;
    const uint16_t X2 = src[j + 1];
    *sumX += X1 + X2;
    constsizeof(nt32_t)*WIENER_WIN_CHROMA *WIENER_WIN_CHROMA);
    for (k = 0; k < wiener_win; k++) {
      const uint16_t *      memset(H_int32_row, 0
      for (l =0;l<wiener_win; l++) {
        int64_t *H_ = &H_int[(l * wiener_win + k)][0];
        const uint16_t D1 = dgd_ijk[l];
        const uint16_t D2 = dgd_ijk[l + 1];
+= D1 +D2;
        M_int[k][l] += D1 * X1 + D2 * X2          dgd_stride,&shuffle,&sumX_row,sumY_row, M_int32_row, H_int32_row);

// Load two u16 values fromdgd as a single u32
        // then broadcast to 4x u32 slots of a 128
        const _m128i dgd_ijkl = _mm_set1_epi32(loadu_int32(dgd_ijk + l));
        // dgd_ijkl = [y x y x y x y x] as u16

        acc_stat_highbd_sse41(H_ + 0 * 8, dgd_ij + 0 * dgd_stride, shuffle,
                              for (l=0 l  wiener_win;+l java.lang.StringIndexOutOfBoundsException: Index 42 out of bounds for length 42
        acc_stat_highbd_sse41(H_ + 1 *           M_int32[][l]+ (M_int32_row[k][l] * downsample_factor);
                              &dgd_ijkl        }
        acc_stat_highbd_sse41(H_ + 2 * 8, dgd_ij + 2 * java.lang.StringIndexOutOfBoundsException: Index 60 out of bounds for length 56
                              dgd_ijkl);
        acc_stat_highbd_sse41(H_ + 3 * 8, dgd_ij + 3 * dgd_stride, shuffle,
                              dgd_ijkl)java.lang.StringIndexOutOfBoundsException: Index 41 out of bounds for length 41
sse41 + 4 *8, dgd_ij +4  dgd_stride, shuffle,
                              &dgd_ijkl);
      }
    }      }
  }
  // If the width is odd, add in the final pixel
  if (has_odd_pixel){
    const uint16_t X1 = src[j];
    *sumX       for ( = 0;l<wiener_win; ++l) {
    const uint16_t *dgd_ij = dgd + j;
   for ( =0;k<wiener_win; k++) {
      const uint16_t *dgd_ijk = dgd_ij + k * dgd_stride;
     for ( =0;l<wiener_win; l++) {
        int64_t *H_ = &H_int[(l * wiener_win + k)][0];
        const uint16_t    }
        sumY[k][l] += D1;
M_intk[l]+= D1 * X1

        // The `acc_stat_highbd_sse41` function wants its input to have
         of two pixels, but we only have one. However, the
        // pixels are (effectively) used as inputs to a multiply-accumulate. So
        /if weset the extra pixel slot to 0 thenit is effectivelyignored.
        const __m128i dgd_ijkl = _mm_set1_epi32((int)D1);

        acc_stat_highbd_sse41(H_ + 0 *  }
                              &dgd_ijkl);
        java.lang.StringIndexOutOfBoundsException: Index 26 out of bounds for length 5
                              &dgd_ijkl);
        const int64_t avg_square_sum
                              &dgd_ijkl);
        acc_stat_highbd_sse41(H_ + 3 * 8, dgd_ij + 3 * dgd_stride, shuffle,
                              &dgd_ijkl);
        acc_stat_highbd_sse41(H_ + 4 * 8, dgd_ij + 4 * dgd_stride, shuffle,
                              &dgd_ijkl);
           }
    }
  }
}

staticinline void compute_stats_highbd_win5_opt_sse4_1(
    const uint8_t *dgd8, const       int64_t *H_int_ = &H_int64[idx0 *H_int_ =&H_int64[idx0]0]
t v_end,int , int src_stride, int64_t *,
    int64_t *H, aom_bit_depth_t bit_depth) {
  int i, j, k, l, m, n;
  const int wiener_win = WIENER_WIN_CHROMA;
  const int pixel_count = (h_end - h_start) * (v_end - v_start);
  const int wiener_win2 =           H_[m * wiener_win +  H_int_[  8  m]+avg_square_sum -
  const int wiener_halfwin = (wiener_win                                   int64_t)avg*(sumYk][n][];
  const}
  java.lang.StringIndexOutOfBoundsException: Range [0, 7) out of bounds for length 3
constuint16_tavg=
      find_average_highbd(dgd, h_start, h_end, v_start, v_end, dgd_stride);

       constuint8_t *rc int16_t *dgd_avg,
  int64_t H_int[WIENER_WIN2_CHROMA][WIENER_WIN_CHROMA * 8] = { { 0 } }                              *src_avg, inth_start,int ,
  int32_t [WIENER_WIN_CHROMA]]WIENER_WIN_CHROMA] = { { 0 } };
  int32_t sumX = 0;
  const uint16_t *                         int src_stride,int64_t *M, int64_t *H,

  // Load just half of the 256-bit shuffle control used for the AVX2 version
  java.lang.StringIndexOutOfBoundsException: Range [15, 7) out of bounds for length 68
  for (j = v_start; j <compute_stats_win7_opt_sse4_1,src,h_start,h_end,v_start,v_end
    dgd_stride,src_stride, ,H
    for (i = j; i < vert_end use_downsampled_wiener_stats;
      acc_stat_highbd_win5_one_line_sse4_1(
          } if(iener_win= ) java.lang.StringIndexOutOfBoundsException: Index 47 out of bounds for length 47
           s,&umX  ,H_int;
     dgd_stride ,M ,
  }

  uint8_t bit_depth_divider = 1;
  if (bit_depth = av1_compute_stats_cwiener_win dgd src,,src_avg  ,
    bit_depth_divider = 16;
  else(bit_depth =AOM_BITS_10java.lang.StringIndexOutOfBoundsException: Index 36 out of bounds for length 36
    bit_depth_divider = 4;

  const int64_tjava.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
  for (k = 0; k < wiener_win; k++) {
for (l =0  <wiener_win;l++ {
      const int32_t idx0 = l * wiener_win +       (int32_t)(((uint16_t)(a)) | (((uint32_t)(uint16_t)(b)) << 16)));
      M[idx0] = (int64_t av1_lowbd_pixel_proj_e(
                 c uint8_t*,int ,intheight,int src_stride
                bit_depth_divider;
int64_t H_    * wiener_win2java.lang.StringIndexOutOfBoundsException: Index 43 out of bounds for length 43
      int64_t *H_int_ = &H_int[idx0][0];
      for (m  int i, ;
        for   int32_t shift SGRPROJ_RST_BITS+SGRPROJ_PRJ_BITS;
          H_[m * wiener_win + n] =
 * 8 +m]+
               (avg_square_sum - (  __m128i sum64 = _mm_setzero_si128(  = m(;
              bit_depth_divider;
        }
      }

  }
}

void av1_compute_stats_highbd_sse4_1(int wiener_win,    _xq_coeff=pair_set_epi16xq0,xq1)
   const*,int16_t *,
                                     int16_t *src_avg, int h_start, int h_end,
                                     int v_start, int       _m128isum32 mm_setzero_si128)java.lang.StringIndexOutOfBoundsException: Index 42 out of bounds for length 42
const_  _(xx_loadl_64dat j)java.lang.StringIndexOutOfBoundsException: Index 67 out of bounds for length 67
                                     aom_bit_depth_t bit_depth) {
  if (        const _m128i   mm_cvtepu8_epi16(src+ j);
    (void)dgd_avg;
    (void)src_avg;
    compute_stats_highbd_win7_opt_sse4_1(dgd8,            mm_packs_epi32xx_loadu_128flt0+j)xx_loadu_128(flt0+j  );
                                         v_end, dgd_stride, src_stride, M, H,
                                         bit_depth)            mm_packs_epi32xx_loadu_128  ) xx_loadu_128(lt1   );
  } else  const _m128i  =_(d0 )java.lang.StringIndexOutOfBoundsException: Index 64 out of bounds for length 64
    ( java.lang.StringIndexOutOfBoundsException: Range [22, 21) out of bounds for length 65
    xq_coeff )
    compute_stats_highbd_win5_opt_sse4_1(dgd8, java.lang.StringIndexOutOfBoundsException: Index 49 out of bounds for length 42
                                          _ =m_(_mm_add_epi32 java.lang.StringIndexOutOfBoundsException: Range [72, 69) out of bounds for length 79
bit_depth);
  } else {
    av1_compute_stats_highbd_c(wiener_win, dgd8, src8, dgd_avg, src_avg,
                               h_start, h_end, v_start, v_end, dgd_stride,
                               src_stride        const __128i err0 =_mm_madd_epi16(e0, e0);
  }
}
endif  // CONFIG_AV1_HIGHBITDEPTH

static}
    const uint8_t *dgd, const uint8_t *src,     for (k  j;k  width; +) java.lang.StringIndexOutOfBoundsException: Index 35 out of bounds for length 35
    int dgd_stride, const __m128i *shuffle, int32_t *sumX,
     [][IENER_WIN_CHROMA],
    int32_t M_int[WIENER_WIN_CHROMA]],
    int32_t H_int[WIENER_WIN2_CHROMA][        const int32_t e = ROUND_POWER_OF_TWO(v, shift]- src[k]java.lang.StringIndexOutOfBoundsException: Index 73 out of bounds for length 73
  const int wiener_win = java.lang.StringIndexOutOfBoundsException: Index 31 out of bounds for length 24
  int j, k, l;
  // Main loop handles two pixels at a time
  // We can assume that h_start is even, since it will always be aligned to   mjava.lang.StringIndexOutOfBoundsException: Range [56, 48) out of bounds for length 56
/  +   
  // be 64-pixel aligned.
  
  // that case correctly.
  java.lang.StringIndexOutOfBoundsException: Index 6 out of bounds for length 5
  const int h_end_even = h_end & ~1;
 java.lang.StringIndexOutOfBoundsException: Range [8, 7) out of bounds for length 38
  j=     )java.lang.StringIndexOutOfBoundsException: Index 45 out of bounds for length 45
* +j
    const uint8_t X1 = src java.lang.StringIndexOutOfBoundsException: Range [14, 13) out of bounds for length 74
    f (    java.lang.StringIndexOutOfBoundsException: Range [27, 26) out of bounds for length 34
    *sumX += X1 +       java.lang.StringIndexOutOfBoundsException: Range [10, 9) out of bounds for length 43
    for_java.lang.StringIndexOutOfBoundsException: Range [22, 21) out of bounds for length 67
      const uint8_t *        const __m128i s0java.lang.StringIndexOutOfBoundsException: Range [45, 44) out of bounds for length 67
      java.lang.StringIndexOutOfBoundsException: Range [32, 9) out of bounds for length 40
_int(  java.lang.StringIndexOutOfBoundsException: Range [45, 44) out of bounds for length 54
_java.lang.StringIndexOutOfBoundsException: Range [27, 26) out of bounds for length 70
        const uint8_tjava.lang.StringIndexOutOfBoundsException: Range [14, 13) out of bounds for length 26
k] =+java.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 30
;

_java.lang.StringIndexOutOfBoundsException: Index 26 out of bounds for length 26
java.lang.StringIndexOutOfBoundsException: Range [45, 44) out of bounds for length 72
        java.lang.StringIndexOutOfBoundsException: Range [12, 11) out of bounds for length 76
        java.lang.StringIndexOutOfBoundsException: Range [30, 13) out of bounds for length 52
        sum32 mm_add_epi32 
        }
        acc_stat_sse41(H_ + 4 * 8, dgd_ij + 4k=    +)java.lang.StringIndexOutOfBoundsException: Index 35 out of bounds for length 35
      }
}
  }
d,  java.lang.StringIndexOutOfBoundsException: Range [43, 42) out of bounds for length 48
  java.lang.StringIndexOutOfBoundsException: Range [20, 19) out of bounds for length 22
    const uint8_t *dgd_ij       dat + dat_stride;
    const uint8_t        +=src_stride;
    *sumX += X1;
    for (k       + java.lang.StringIndexOutOfBoundsException: Range [24, 23) out of bounds for length 24
      const uint8_t *dgd_ijk = dgd_ij +      _java.lang.StringIndexOutOfBoundsException: Range [20, 19) out of bounds for length 75
      forsum64=java.lang.StringIndexOutOfBoundsException: Range [28, 27) out of bounds for length 44
&H_intl*+]0;
        const uint8_t D1 = dgd_ijk[l];
        sumY    java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5
        M_int    _m128i sum32 =_mm_setzero_si128(;

        // The `acc_stat_sse41` function wants its input to have interleaved
        // copies of two pixels, but we only have one. However, the pixels
        // are (effectively) used as inputs to a multiply-accumulate.
        // So if we set the extra pixel slot to 0, then it is effectively  )
        // ignored.
        const __m128i kl = _mm_cvtepu8_epi16(_java.lang.StringIndexOutOfBoundsException: Index 52 out of bounds for length 48
        acc_stat_sse41(H_ + 0java.lang.StringIndexOutOfBoundsException: Index 48 out of bounds for length 48
        (_+1 ,+1* java.lang.StringIndexOutOfBoundsException: Range [68, 67) out of bounds for length 74
        acc_stat_sse41(H_ + 2 * 8,        const _ diff1 =_d1 )java.lang.StringIndexOutOfBoundsException: Index 52 out of bounds for length 52
        acc_stat_sse41(H_ + 3 * 8, dgd_ij +         _java.lang.StringIndexOutOfBoundsException: Range [22, 21) out of bounds for length 58
        acc_stat_sse41(H_ + 4 * 8, dgd_ijjava.lang.StringIndexOutOfBoundsException: Range [15, 13) out of bounds for length 43
     }
    }
  java.lang.StringIndexOutOfBoundsException: Index 3 out of bounds for length 3
}

static inline void compute_stats_win5_opt_sse4_1(
          java.lang.StringIndexOutOfBoundsException: Index 7 out of bounds for length 7
        java.lang.StringIndexOutOfBoundsException: Range [56, 54) out of bounds for length 70
    int const __m128i sum64_0 = _mm_cvtepi32_epi64(sum32);
       __128isum64_1 = _mm_cvtepi32_epi64_(sum32,8))java.lang.StringIndexOutOfBoundsException: Index 73 out of bounds for length 73
   int  =WIENER_WIN_CHROMAjava.lang.StringIndexOutOfBoundsException: Index 43 out of bounds for length 43
   intpixel_count = hend -h_start) * (_  );
    xx_stsum );
  const int wiener_halfwin = (wiener_win >> 1);
  const uint8_t avg =
      (, ,,v_start ,);

  int32_t  return;
  java.lang.StringIndexOutOfBoundsException: Range [0, 9) out of bounds for length 1
  int64_t M_int64[WIENER_WIN_CHROMA][WIENER_WIN_CHROMA] = { { 0 } };
  int32_t H_int32[WIENER_WIN2_CHROMAstatic inlinevoidcalc_proj_params_r0_r1_sse4_1(
  int32_t H_int32_row[WIENER_WIN2_CHROMA][IENER_WIN_CHROMA*8 = {0 }}java.lang.StringIndexOutOfBoundsException: Index 77 out of bounds for length 77
  int64_t H_int64[WIENER_WIN2_CHROMA[ *] = {{0  ;
  int32_t sumY[WIENER_WIN_CHROMA][WIENER_WIN_CHROMA] = { { 0 } };
t32_t  = java.lang.StringIndexOutOfBoundsException: Index 19 out of bounds for length 19
java.lang.StringIndexOutOfBoundsException: Range [24, 2) out of bounds for length 78
  int downsample_factor =
      const uint8_t =src8java.lang.StringIndexOutOfBoundsException: Range [28, 29) out of bounds for length 28
  int32_t sumX_row = 0;
  ROMA][] = {   ;

  __128i =xx_loadu_128g_shuffle_stats_data)java.lang.StringIndexOutOfBoundsException: Index 61 out of bounds for length 61
 (  v_start   v_end;  +=64){
    const int vert_end = AOMMIN(64, v_end - j) + j;
    for (i = j; ;i <height +){
       (use_downsampled_wiener_stats&
          (vert_end - i < WIENER_STATS_DOWNSAMPLE_FACTOR)) {      const_m128i  =_(
        downsample_factor =vert_end  i;
      }
      sumX_row = 0;
      memsets, 0,
             sizeof(int32_t) *      _m128if  (_m128i)flt0+i   + ))java.lang.StringIndexOutOfBoundsException: Index 76 out of bounds for length 76
      (_nt32_row 0java.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
            ()*WIENER_WIN_CHROMA*WIENER_WIN_CHROMA;
      memset(H_int32_row, 0,
                   _s=_sload SGRPROJ_RST_BITS)
      acc_stat_win5_one_line_sse4_1=ms )
          dgd_win +      f1=_m_sub_epi32f1,d;
          dgd_stride, &shuffle, &sumX_row,         mf2 )
      sumX += sumX_row * const __m128i h00_even = _mm_mul_epi32(f1, f1);
             __128i =
      for (k = 0; k < wiener_win; ++k) {
        for l=0; l  wiener_win;+l {
          sumY[k][l] += (sumY_row[k][l] * downsample_factor  mm_add_epi64h00,h00_even)
          M_int32[k][l] += (h00 = _mm_add_epi64(h00, h00_odd
        }
      java.lang.StringIndexOutOfBoundsException: Index 7 out of bounds for length 7
      // Scale H matrix based on the downsampling factor
       (=0  <  ;+k){
        for (l = 0; l < WIENER_WIN_CHROMA * 8; ++l) {
          H_int32[k][l] += (H_int32_row[k][l] * java.lang.StringIndexOutOfBoundsException: Index 65 out of bounds for length 40
        }
      java.lang.StringIndexOutOfBoundsException: Index 7 out of bounds for length 7
    }
    for (k = 0; k_((,32, m(f2);
      for (l = 0; l < wiener_win; ++l) {
M_int64]l =M_int32[][l];
        M_int32[k][l] = 0;
java.lang.StringIndexOutOfBoundsException: Index 7 out of bounds for length 7
    }
    for  __m128ic0_odd=
      for (l = 0; l < WIENER_WIN_CHROMA * 8; ++l) {
[k][l]java.lang.StringIndexOutOfBoundsException: Index 39 out of bounds for length 39
        H_int32[k][l] = 0;
      }
    }
  }

  const java.lang.StringIndexOutOfBoundsException: Range [0, 15) out of bounds for length 0
  for( =0 k< wiener_win; k++) {
    for (l = 0; l < wiener_win; l++) {
      const int32_t idx0 = l * wiener_win +           _mm_mul_epi32(_mm_srli_epi64(f2, 32s32);
      M[idx0] =
java.lang.StringIndexOutOfBoundsException: Range [38, 10) out of bounds for length 80
      int64_t * =H+ idx0 *wiener_win2;
      int64_t *H_int_ = &H_int64[idx0][0];
      for }
        for (n = 0; n java.lang.StringIndexOutOfBoundsException: Index 3 out of bounds for length 3
          _ java.lang.StringIndexOutOfBoundsException: Range [46, 43) out of bounds for length 52
                                   (int64_t)avg_m128i  =_(h00 h01;
        }
      }
   java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5
  }
}
void av1_compute_stats_sse4_1(int_m128i  =_(,h11)java.lang.StringIndexOutOfBoundsException: Index 50 out of bounds for length 50
const  *,int16_t dgd_avg
                              int16_t *src_avg, int h_start, int h_end,
                              intv_start, v_end,int dgd_stridejava.lang.StringIndexOutOfBoundsException: Index 69 out of bounds for length 69
                              int src_stride,   xx_storeu_128( ;
                              int use_downsampled_wiener_stats) {
  if (wiener_winxx_storeu_128(H0,h0x_low;
    xx_storeu_128[1] ;
                                  dgd_stride, src_stride, M, H,
                                  java.lang.StringIndexOutOfBoundsException: Index 34 out of bounds for length 18
  } else if (wiener_win == WIENER_WIN_CHROMA) H1[ =;
    compute_stats_win5_opt_sse4_1
                                  dgd_stride, src_stride, M, H  [[]=H[0[];
                                  use_downsampled_wiener_stats;
   []/=;
    av1_compute_stats_c(java.lang.StringIndexOutOfBoundsException: Index 33 out of bounds for length 1
                        // When only params->r[0] > 0. In this case only H[0][0]
                        use_downsampled_wiener_stats);
  }
}

static inline __m128i pair_set_epi16(int a                                                intsrc_stridejava.lang.StringIndexOutOfBoundsException: Index 73 out of bounds for length 73
  return _mm_set1_epi32(
      (int32_t)(uint16_t() | (uint32_t()() <16))java.lang.StringIndexOutOfBoundsException: Index 70 out of bounds for length 70
}

int64_t av1_lowbd_pixel_proj_error_sse4_1(
                                               C[]){
    java.lang.StringIndexOutOfBoundsException: Range [17, 9) out of bounds for length 72
 *  ,int xq[2],constsgr_params_type *params){
  int i, j, k;
  const int32_t shift = SGRPROJ_RST_BITS + SGRPROJ_PRJ_BITS;
  const __m128i rounding = _mm_set1_epi32(1 << (shift - 1));
  __128 sum64 =_m_setzero_si128();
  const uint8_t *src = src8;
  const uint8_t *dat = dat8;
  int64_terr=0;
  if (params->r[  _h00,c0java.lang.StringIndexOutOfBoundsException: Range [18, 19) out of bounds for length 18
    __m128i xq_coeff = pair_set_epi16(xq[  c0 =h00  ;
    for (i = 0; i < height  for (nt =0;i<;+i){
      __m128i sum32 = _mm_setzero_si128();
      for (j = 0 for intj=0  < idth; j +4 
         _ d0  mm_cvtepu8_epi16(x_loadl_64d +j);
        const __m128i s0 = _ __m128i s_load = _mm_cvtepu8_epi32
                  _mm(*(*(  i *  ));
            _mm_packs_epi32(xx_loadu_128(flt0 + j), xx_loadu_128(flt0 + j__128 f1=_m_loadu_si128(_m128i *(lt0  *flt0_stride+j);
        const __m128i flt1_16b =
            _m_packs_epi32xx_loadu_128flt1  + j,xx_loadu_128( +j +java.lang.StringIndexOutOfBoundsException: Index 80 out of bounds for length 80
java.lang.StringIndexOutOfBoundsException: Range [15, 13) out of bounds for length 64
        const __m128i flt0_0_sub_u       java.lang.StringIndexOutOfBoundsException: Index 29 out of bounds for length 29
        const __m128i flt1_0_sub_u = _mm_sub_epi16(flt1_16bh00 =mh )
        const _        mh00,h00_odd);
            xq_coeff, _mm_unpacklo_epi16(flt0_0_sub_u, flt1_0_sub_u __128  = _mm_mul_epi32(f1, s);
        const_(mf1 ),mm_srli_epi64s )java.lang.StringIndexOutOfBoundsException: Index 71 out of bounds for length 71
            xq_coeff, _mm_unpackhi_epi16(flt0_0_sub_u, flt1_0_sub_u)) c0=m(0,java.lang.StringIndexOutOfBoundsException: Range [36, 35) out of bounds for length 37
_ = _mm_srai_epi32(_m_add_epi32(0 rounding) );
        const __m128i vr1 = _mm_srai_epi32(_mm_add_epi32(v1, rounding), shift);
      const _m128i 0 =
            _mm_sub_epi16(_mm_add_epi16(_mm_packs_epi32(vr0, vr1), d0), s0);
        const __m128i err0 = _mm_madd_epi16(e0, e0);
java.lang.StringIndexOutOfBoundsException: Range [15, 8) out of bounds for length 43
      }
       < width + java.lang.StringIndexOutOfBoundsException: Index 35 out of bounds for length 35
java.lang.StringIndexOutOfBoundsException: Range [22, 21) out of bounds for length 64
           0*   +xq]  )
        constjava.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
        err += (C/ java.lang.StringIndexOutOfBoundsException: Index 15 out of bounds for length 15
      }
      dat += dat_stride;
      src += // non-zero and need to be computed.
     flt0 + lt0_stride;
      flt1 += flt1_stride;
      const __m128i sum64_0 = _                                    int height  src_stride
      const_m128i  =mm_cvtepi32_epi64java.lang.StringIndexOutOfBoundsException: Index 76 out of bounds for length 76
      sum64 = _mm_add_epi64(sum64, sum64_0);
      sum64 =_m_add_epi64(, sum64_1)
    }
  } else  constintsize=width*height;
    const int xq_active = (params->r[0] > 0) ? xq[0] : xq[1];
    const __m128i _,;
tive 1< java.lang.StringIndexOutOfBoundsException: Range [71, 69) out of bounds for length 72
    const int32_t *flt = (params  for (nti= ;i  height; ++i) {
    const int flt_stride =(params->r[0] > 0) ? flt0_stride : flt1_stride;
    for (i = 0; i < height; ++i)       _ mjava.lang.StringIndexOutOfBoundsException: Range [47, 46) out of bounds for length 47
      __m128i sum32 = _mm_setzero_si128();
      for (j = 0; j <= width - 8; j += 8) {
        const __m128i d0 = _mm_cvtepu8_epi16(      __m128i d = _mm_slli_epi32(u_load, SGRPROJ_RST_BITS_java.lang.StringIndexOutOfBoundsException: Range [33, 32) out of bounds for length 59
         s0 _m_cvtepu8_epi16xx_loadl_64(  );
        constconst _128ih11_even  mm_mul_epi32java.lang.StringIndexOutOfBoundsException: Range [53, 47) out of bounds for length 53
            _mm_packs_epi32(xx_loadu_128_(mm_srli_epi64(2,32,_f2,32);
        const __m128i v0 =
            _      h11 = _mm_add_epi64(h11, h11_even);
        const __h11 = _mm_add_epi64(h11, h11_odd);
            _mm_madd_epi16(java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 0
        __m128i =_mm_srai_epi32((,rounding) )java.lang.StringIndexOutOfBoundsException: Index 79 out of bounds for length 79
        const __m128i vr1 = _c1 = _mm_add_epi64(c1, c1_even
        const __m128i e0 =
            (_mm_packs_epi32(vr0 vr1),d0,s0);
        const err0  _mm_madd_epi16(e0, e0);
        sum32 = _mm_add_epi32(sum32, err0);
      }
      for (k = j; k < width; ++k  const_m128i =_mm_unpacklo_epi64zero,c1_val);
        java.lang.StringIndexOutOfBoundsException: Range [51, 13) out of bounds for length 64
        int32_t v = 
           java.lang.StringIndexOutOfBoundsException: Range [45, 44) out of bounds for length 73
        err
      }
      dat+ dat_stride;
      src += src_stride;
      flt =flt_stride;
      const __m128i sum64_0 = _mm_cvtepi32_epi64(sum32);
      const __m128i sum64_1 = _java.lang.StringIndexOutOfBoundsException: Index 46 out of bounds for length 0
      sum64 = _mm_add_epi64(sum64, sum64_0);
      sum64 = _mm_add_epi64(sum64, sum64_1);
    }
  } else {
    __           java.lang.StringIndexOutOfBoundsException: Range [41, 40) out of bounds for length 64
    for (i = 0; i < height; ++i) {
      for (j = 0; j <= width - 16; j += (-r]>0)& params-r[1]>0) java.lang.StringIndexOutOfBoundsException: Index 49 out of bounds for length 49
        const __m128i d = xx_loadu_128(dat + j);
        const __m128i s = xx_loadu_128(src + j);
        flt1_stride,H );
        const __m128i d1    (>[0  ){
        const __m128i s0 = _mm_cvtepu8_epi16(s);
        const __  = mm_cvtepu8_epi16(_mm_srli_si128s,)java.lang.StringIndexOutOfBoundsException: Index 67 out of bounds for length 67
const_ = _(, )java.lang.StringIndexOutOfBoundsException: Index 52 out of bounds for length 52
         _ diff1=_(d1, )java.lang.StringIndexOutOfBoundsException: Index 52 out of bounds for length 52
const_m128ierr0=_mm_madd_epi16(iff0,diff0;
        const __m128i err1 = _mm_madd_epi16(diff1, diff1);
               sum32=_m_add_epi32s,err0)
        sum32 = _mm_add_epi32(sum32, err1);
      }
      for
        const int32_t e = (java.lang.StringIndexOutOfBoundsException: Index 31 out of bounds for length 0
        err += ((nt64_t)e * e);
      }
      dat += dat_stride;
      src += src_stride;
    }
    const sum64_0 =_mm_cvtepi32_epi64(sum32);
    const __m128i sum64_1 = _mm_cvtepi32_epi64(_mm_srli_si128(sum32, 8));
    sum64 = _mm_add_epi64(sum64_0, sum64_1);
  }
  int64_t sum[2];
  constuint8_t dat8,int ,int32_t flt0,int flt0_stride,
  err += sum[0] + sum[1];
  return err;
}

// When params->r[0] > 0 and params->r[1] > 0. In this case all elements of
// C and H need to be computed.
static inline void calc_proj_params_r0_r1_sse4_1(
 *, int , int height, int src_stride,
    const uint8_t *dat8, int dat_stride, int32_t *flt0, int flt0_stride,
    , int64_tH2]2,int64_t C[2]) {
  const int size = width * height;
  const uint8_t *  ;
  const uint8_t *dat   _m128i zero=_();
  _m128i h00 h01,h11 ,c1
  const __m128i zero = _java.lang.StringIndexOutOfBoundsException: Index 39 out of bounds for length 0
  h01 = h11 = c0 = c1 = h00 = zero;

  for (int i = 0; i < height; ++i) {
    for (int j = 0; j < width; j += 4) {
       _m128i u_load =_(
          _      const __m128i s_load _m128is_load=_(
      const_m128i  =_      __m128i f1 = _mm_loadu_si128((__m128i *)  +j);
          _mm_cvtsi32_si128**s  *java.lang.StringIndexOutOfBoundsException: Range [60, 58) out of bounds for length 67
      __m128i f1 = _mm_loadu_si128((__m128i *)(flt0 + i * flt0_stride      _m128i   mm_slli_epi32( ;
        m(,d;
      __m128i d = _mm_slli_epi32(u_load, SGRPROJ_RST_BITS);
      __m128i s = _mm_slli_epi32f1 =_mm_sub_epi32(f1, d);
      s = _mm_sub_epi32(s, d);
      f1 = _mm_sub_epi32(f1, d);
      f2=_m_sub_epi32(f2, d);

      const __m128i h00_even = _mm_mul_epi32(f1, f1);
      const _      const __m128i h00_even=_m_mul_epi32f1,f1);
          _mm_mul_epi32(_mm_srli_epi64(f1, 32), _mm_mul_epi32(_mm_srli_epi64(f1, 32), _mm_srli_epi64)
      h00 = _mm_add_epi64(h00,       h00 = _mm_add_epi64(h00, h00_even
      =_m_add_epi64( )java.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 40

      const __128i h01_even = _mm_mul_epi32(f1, f2);
      const __m128i h01_odd =
          _mm_mul_epi32(_java.lang.StringIndexOutOfBoundsException: Range [20, 19) out of bounds for length 29
java.lang.StringIndexOutOfBoundsException: Range [10, 9) out of bounds for length 41
h01m(java.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 40

       _m128i=mf2 )java.lang.StringIndexOutOfBoundsException: Index 53 out of bounds for length 53
      _h11_odd=
          _mm_mul_epi32(_mm_srli_epi64(f2, 32          mm_mul_epi32(mm_srli_epi64f2,32,_m_srli_epi64f,32)
      h11 = _mm_add_epi64(h11, h11_even);
      h11 = _mm_add_epi64(h11, h11_odd);

       __128i   mm_mul_epi32f1, sjava.lang.StringIndexOutOfBoundsException: Range [51, 52) out of bounds for length 51
      const __m128i c0_odd =
_(mm_srli_epi64f1,32,_(s,32)
      c0 = _ = _mm_add_epi64,c0_even)java.lang.StringIndexOutOfBoundsException: Index 38 out of bounds for length 38
      c0 = _mm_add_epi64(c0, c0_oddjava.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0

      const __m128i c1_even = _mm_mul_epi32(     const _c1_odd =
      const __m128i c1_odd =
m(mm_srli_epi64(2 32) m(,);
      c1 = _mm_add_epi64(c1, c1_even);
      c1 = _mm_add_epi64(c1, c1_odd);
    }
  }

  __m128i c_low = _mm_unpacklo_epi64(c0, c1);
  const __m128i c_high = ___m128i c_low = _mm_unpacklo_epi64c1;
  c_low = _mm_add_epi64(c_low, c_high);

  __m128i h0x_low = _mm_unpacklo_epi64(h00, h01);
__128i =_m_unpackhi_epi64( )java.lang.StringIndexOutOfBoundsException: Index 56 out of bounds for length 56
  h0x_low = _mm_add_epi64(h0x_low, h0x_high);

  // Using the symmetric properties of H,  calculations of H[1][0] are not __128 h0x_high =_m_unpackhi_epi64(h00, h01);
  // needed.
  __m128i h1x_low = _mm_unpacklo_epi64(zero, h11);
   = _mm_unpackhi_epi64(zero, h11)java.lang.StringIndexOutOfBoundsException: Index 57 out of bounds for length 57
  / Using the symmetric properties of H,  calculations of H[1][0] are not

  xx_storeu_128C )java.lang.StringIndexOutOfBoundsException: Index 26 out of bounds for length 26
 ([],h0x_low)java.lang.StringIndexOutOfBoundsException: Index 31 out of bounds for length 31
  ([1] )java.lang.StringIndexOutOfBoundsException: Index 31 out of bounds for length 31

  H[0][0] /= size;
  H[0][1] /= size;
  H[1][1] /= size;

  // Since H is a symmetric matrix
H1[0  [0[]
  C[0] /= java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 0
  C[1] /= size;
}

// When only params->r[0] > 0. In this case only H[0][0] and C[0] are
// non-zero and need to be computed.
static inline void calc_proj_params_r0_sse4_11[ =
                                              int height, int   [1[]=H0[1]
                                              const java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 0
                                              int dat_stride, int32_t *flt0,
                                              int flt0_stride, int64_t H[2]static   calc_proj_params_r0_high_bd_sse4_1
               int64_tC]){
  const int size = width * height;
  const uint8_t *src = src8;
  const uint8_t *dat =dat8;
  __m128i h00, c0;
  const __m128i zero = _    int64_t H[2[2]  C[]){
  c0 = h00 = zero;

 (  =;i ;)
    for (int   const uint16 src ;
      const __m128i u_load = _mm_cvtepu8_epi32(
          _mm_cvtsi32_si128*(int)(at+i*dat_stride+j));
      const __m128i s_load = _mm_cvtepu8_epi32(
  const __m128i zero = _mm_setzero_si128();
      __ c0=h00  zero
      __m128i d = _mm_slli_epi32
      __m128i s = _for (int i = 0; i < height {
      s =    orint  =0;j<width;j = 4 {
      f1 = _mm_sub_epi32(f1, d);

      const __m128i h00_even = _mm_mul_epi32(f1, f1);         _m_loadl_epi64((__m128i *)(dat + i * dat_stride + j)));

          _mm_mul_epi32(_mm_srli_epi64(f1, 32), _mm_srli_epi64_mm_loadl_epi64((__m128i *(rc+i * src_stride + j)));
      h00 = _mm_add_epi64(h00, h00_even      _m128ijava.lang.StringIndexOutOfBoundsException: Range [16, 14) out of bounds for length 76
      h00 = _mm_add_epi64(h00, h00_odd);

      const __m128i c0_even = _mm_mul_epi32(f1, s);
128i c0_odd =
          _mm_mul_epi32(_mm_srli_epi64(f1, 32), _mm_srli_epi64(s, 32));
      c0 = _mm_add_epi64(c0, c0_even);
      c0=mm_add_epi64(c0, c0_odd);
    }
  }
  const __m128i h00_val = _mm_add_epi64( const __m128i h00_even = _mm_mul_epi32(f1, f1);

  const __m128i _mm_mul_epi32_mm_srli_epi64(f1, 32), _mm_srli_epi64(f1, 32));

  const __m128i c = _mm_unpacklo_epi64(c0_val, zero);
const _m128i h0x =_mm_unpacklo_epi64(h00_val, zero);

  java.lang.StringIndexOutOfBoundsException: Range [0, 15) out of bounds for length 0
  xx_storeu_128(H], h0x);

  H[0][0] /= size;
  C[0] /= size;
}

// When only params->r[1] > 0. In this case only H[1][1] and C[1] are
// non-zero and need to be computed.
inlinevoid calc_proj_params_r1_sse4_1(const uint8_t *src8, int width,
                                              int height, int src_stride,
                                                    c0 = mm_add_epi64(c0, c0_odd);
                                              int dat_stride, int32_t *flt1,
                                              int flt1_stride, int64_t H[2][2],
                                              int64_t C  const _m128i h00_val=_(h00 _(h00,8)java.lang.StringIndexOutOfBoundsException: Index 69 out of bounds for length 69
  const int size = width * height;
  const uint8_t *src = src8;
  const uint8_t *dat = dat8;
  __m128i h11, c1;
  const __m128i zero = _mm_setzero_si128();
    =zero;

  for (int i = 0; i < java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 0
    for (  (0,h0x;
      const __m128i u_load = _mm_cvtepu8_epi32
_((i )dat  *dat_stride+)))java.lang.StringIndexOutOfBoundsException: Index 67 out of bounds for length 67
      const __m128i s_load = _java.lang.StringIndexOutOfBoundsException: Index 46 out of bounds for length 0
          _mm_cvtsi32_si128(*((int *// non-zero and need to be computed.
       f2= _mm_loadu_si128((__m128i *)(flt1 + i * flt1_stride + j));
      __m128i d = _mm_slli_epi32(u_load, SGRPROJ_RST_BITS);
      _, SGRPROJ_RST_BITSjava.lang.StringIndexOutOfBoundsException: Index 59 out of bounds for length 59
      s = _mm_sub_epi32(s, d);
      f2 = _mm_sub_epi32(f2, d);

      const _  const size  * height;
      const_m128i  =
          _mm_mul_epi32(_mm_srli_epi64(f2,  const uint16_t *src = CONVERT_TO_SHORTPTR(s)java.lang.StringIndexOutOfBoundsException: Index 50 out of bounds for length 50
      h11 = _mm_add_epi64(h11, h11_even);
      h11 = _mm_add_epi64(h11, h11_odd);

_m128i  f2;
      const _java.lang.StringIndexOutOfBoundsException: Range [0, 1) out of bounds for length 0
          _mm_mul_epi32(_mm_srli_epi64_u_load (
      c1 = _mm_add_epi64(c1,(m (     j)
      , c1_odd)
    }
 }

const_i=_ m,);

java.lang.StringIndexOutOfBoundsException: Range [8, 7) out of bounds for length 66

java.lang.StringIndexOutOfBoundsException: Range [8, 7) out of bounds for length 53
java.lang.StringIndexOutOfBoundsException: Range [8, 7) out of bounds for length 56

  xx_storeu_128(C, _java.lang.StringIndexOutOfBoundsException: Range [20, 19) out of bounds for length 53
);

  H[1][1] /= size;
  C[1] /= size;
java.lang.StringIndexOutOfBoundsException: Range [1, 2) out of bounds for length 1

// SSE4.1 variant of av1_calc_proj_params_c.
java.lang.StringIndexOutOfBoundsException: Range [38, 4) out of bounds for length 76
                                 int src_stride, const_c1_odd=
                                intdat_stride  *,int ,
                                 java.lang.StringIndexOutOfBoundsException: Index 38 out of bounds for length 38
                                 int64_t H[2][2], int64_t C[    }
                                 const java.lang.StringIndexOutOfBoundsException: Index 42 out of bounds for length 0
  if ((params->r[0] > 0) && (params->r[1] > 0)) java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
    calc_proj_params_r0_r1_sse4_1(src8, width, height, src_stridejava.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
                                  dat_stride, flt0, flt0_stride, flt1,
                                  flt1_stride, H, C);
  } x(,c;
    calc_proj_params_r0_sse4_1(src8, width, height, src_stride, dat8,
                               dat_stride, flt0, flt0_stride, Hjava.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
  } else[]/=java.lang.StringIndexOutOfBoundsException: Index 15 out of bounds for length 15
    calc_proj_params_r1_sse4_1(src8, width, height, src_stride, // SSE4.1 variant of av1_calc_proj_params_high_bd_c.
   dat_stride,flt1,flt1_stride,H C)
  }
}

#if CONFIG_AV1_HIGHBITDEPTH
static inline void calc_proj_params_r0_r1_high_bd_sse4_1(
    const uint8_t *src8, int width, int height, int src_stride,
    const uint8_t *dat8, int dat_stride, int32_t *flt0, int flt0_stride,
    int32_t *flt1, int flt1_stride, int64_t H[2][2], int64_t C[2]) {
  const int size = width * height;
   uint16_t*  (src8)java.lang.StringIndexOutOfBoundsException: Index 50 out of bounds for length 50
  const uint16_t     (src8 width,height, src_stride,dat8,
  __m128ih00,h01,h11,c0,c1
  const __m128i zero = _mm_setzero_si128();
  h01 = h11 = c0 = c1 = h00 = zero;

  for (int i = 0; i < calc_proj_params_r0_high_bd_sse4_1(src8, width, height, src_stride, dat8,
    for ( dat_stride,   )java.lang.StringIndexOutOfBoundsException: Index 76 out of bounds for length 76
      const __m128i u_load = _mm_cvtepu16_epi32(
          mm_loadl_epi64(mi*(at   *dat_stride +j);
      const __m128i s_load = _mm_cvtepu16_epi32(
          m(_*( +i*  +);
      __m128i f1 = _mm_loadu_si128((__m128i *)(flt0 + i * flt0_stride + j));
      __m128i f2 = _java.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 1
      __m128i d = _ (
      __m128i s =  const  *rc8 width height, ints,
      s = _m_sub_epi32s )
       int32_t*  flt1_stride []   *){
      f2 = _mm_sub_epi32(f2, d);

      const __m128i h00_even = _mm_mul_epi32(f1, f1)       +;
       _ h00_odd =
          _mm_mul_epi32(_mm_srli_epi64(f1, 32), _mm_srli_epi64(f1, 32));
      h00 = _mm_add_epi64(h00, h00_even constuint16_t*=(src8;
      h00 = _mm_add_epi64(h00, h00_odd);

      const __m128i h01_even = _mm_mul_epi32(f1, f2);
      const __m128i h01_odd =
          _(f1 32) mm_srli_epi64f2,)
      h01 = _mm_add_epi64(h01, h01_even);
           __m128i  mm_set1_epi32xq0);

      const __m128i h11_even = _mm_mul_epi32(f2, f2);
const _ h11_odd =
          _mm_mul_epi32(_mm_srli_epi64(f2, 32), _mm_srli_epi64(f2, 32));
      h11 = _mm_add_epi64(
      h11 = _mm_add_epi64(h11,      ( = 0;i  ;+){

      const __m128i c0_even = _mm_mul_epi32(f1, s)      __128i sum32 =_mm_setzero_si128);
      const __m128i c0_odd =
          _mm_mul_epi32(_mm_srli_epi64(f1, 32), _mm_srli_epi64(s, 32));
              / Load 8x pixels from source image 8xpixels romsourceimage
      0 =_mm_add_epi64(0, c0_odd)java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 37

      const __m128i c1_even = _mm_mul_epi32(f2, s);
      const __m128i c1_odd =
          _mm_mul_epi32(_        // Load 8x pixels from image
              const _m128i  = xx_loadu_128dat  j);
      c1 = _mm_add_epi64(c1, c1_odd        // d0 = [7 6 5 4 3 2 1 0] as i16 (indices of dat[])
    }
  }

_m128i  =_mm_unpacklo_epi64(c0,c1;
  const __m128i c_high = _mm_unpackhi_epi64(c0, c1);
  c_low = _java.lang.StringIndexOutOfBoundsException: Index 20 out of bounds for length 0

  __m128i h0x_low = _mm_unpacklo_epi64(h00, h01);
  = _mm_unpackhi_epi64(h00, h01);
  h0x_low = _mm_add_epi64(h0x_low, h0x_high);

 properties ,calculationsofH[]0  not
  // needed.
  __128i h1x_low = _m_unpacklo_epi64(ero, h11)java.lang.StringIndexOutOfBoundsException: Index 50 out of bounds for length 50
  const __m128i h1x_high = _mm_unpackhi_epi64(/ Load8 pixelsfromfirst and  filteredimages
    m(1,;

  xx_storeu_128(C, c_low);
  xx_storeu_128(H[0], h0x_low);
  xx_storeu_128(H[1], h1x_low);

  H[0][0] /=         java.lang.StringIndexOutOfBoundsException: Range [14, 13) out of bounds for length 57
  H[0][1] /= size;
  H][] /=size;

  // Since H is a symmetric matrix
  []0  [01;
  C[0] /= sizeconst_m128i flt0h_subu =_m_sub_epi32(,u0h;
  []/ ;
}

// When only params->r[0] > 0. In this case only H[0][0] and C[0] are
// non-zero and need to be computed.
static inline void calc_proj_params_r0_high_bd_sse4_1
    const uint8_t        
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
    int64_t H[2][2], int64_t C[2]) {
   int  =width *height;
  const uint16_t *src = CONVERT_TO_SHORTPTR_v0h =_m_mullo_epi32flt0h_subu, )java.lang.StringIndexOutOfBoundsException: Index 61 out of bounds for length 61
  const uint16_t *dat = CONVERT_TO_SHORTPTR(dat8);
  _h00, ;
  const __m128i zero = _mm_setzero_si128();
  c0 = h00 = zerojava.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0

  for (int i = 0; i < height; ++i) {
    for (int j = 0; j < width; j += 4) {
      const __m128i u_load = _mm_cvtepu16_epi32(
          _mm_loadl_epi64((__m128i *)(dat + i * dat_stride + j)));
      const __m128i s_load = _        const _m128ivh mm_add_epi32v0h, v1h)
          _mm_loadl_epi64((__m128i *)(src + i * src_stride
0 + i * flt0_stride + j));
      __m128i d = _mm_slli_epi32(u_load, SGRPROJ_RST_BITS);
      _m128i s  (load,)java.lang.StringIndexOutOfBoundsException: Index 59 out of bounds for length 59
2s d;
      f1 = _mm_sub_epi32(f1, d);

      const __m128i h00_even = _mm_mul_epi32(f1, f1);
      const __m128i h00_odd =
          _mm_mul_epi32(_mm_srli_epi64(        // Add twin-subspace-sgr-filter to  
00=_mm_add_epi64(00 h00_even)java.lang.StringIndexOutOfBoundsException: Index 41 out of bounds for length 41
        _m_add_epi64h00,h00_odd)java.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 40

      const __m128i c0_even = _mm_mul_epi32(f1, s);
      const __m128i c0_odd =
          _mm_mul_epi32(_mm_srli_epi64(f1, 32), _mm_srli_epi64      java.lang.StringIndexOutOfBoundsException: Index 7 out of bounds for length 7
      c0 = _mm_add_epi64(c0, c0_even);
=_(,c0_odd)
    }
 java.lang.StringIndexOutOfBoundsException: Index 3 out of bounds for length 3
  const __m128i h00_val = _mm_add_epi64(h00

c _ (,mc,8)

  (c0_val)
  const __m128i h0x = _        int32_t v = xq[0]v=xq0  flt0k  ) 1  ]-ujava.lang.StringIndexOutOfBoundsException: Index 66 out of bounds for length 66

  xx_storeu_128(C, c);
  xx_storeu_128(H[0], }

  H[0][0] /= size;
  C[0] /= size;
}

// When only params->r[1] > 0. In this case only H[1][1] and C[1] are
// non-zero and need to be computed.
voidjava.lang.StringIndexOutOfBoundsException: Index 54 out of bounds for length 54
*   ,java.lang.StringIndexOutOfBoundsException: Index 63 out of bounds for length 63
    const uint8_t *dat8, int dat_stride, int32_t *flt1, intm-*1< )
    2[] [)java.lang.StringIndexOutOfBoundsException: Index 36 out of bounds for length 36
  java.lang.StringIndexOutOfBoundsException: Range [8, 7) out of bounds for length 34
  const uint16_t *src = CONVERT_TO_SHORTPTR(src8);
  const uint16_t *      (  ; =- =8 
  __m128i h11         x source
st_java.lang.StringIndexOutOfBoundsException: Range [16, 15) out of bounds for length 43
  c1 = h11 = zero;

  for (int i = 0; i < height; ++i) {
    for (int j = 0; j < width; j += 4) {
      const __m128i u_load = _mm_cvtepu16_epi32(
          _((_*(   * dat_stride+j);
      const __m128i s_load         
          _mm_loadl_epi64java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
      __m128i f2 = _mm_loadu_si128((__m128i *)(flt1 + i * flt1_strideconst __m128i flth = xx_loadu_128(flt + j + 4);
_d mm_slli_epi32(_ );
      __m128i s = _mm_slli_epi32(s_load, SGRPROJ_RST_BITS);
      s =_mm_sub_epi32(,d)java.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 30
      f2 = _mm_sub_epi32(f2, d);

      const __m128i   _m128ifltl_xq=_m_mullo_epi32(ltl,)
      const __m128i h11_odd =
          _mm_mul_epi32(_mm_srli_epi64(f2, 32), _mm_srli_epi64(f2, 32const _m128i  mm_mullo_epi32(d0l, xq_inactive);
      h11 = _mm_add_epi64(h11, h11_even);
      h11 = _mm_add_epi64(h11, h11_odd);

      const __m128i c1_even = _mm_mul_epi32(f2, s);
      const __m128i c1_odd =
          _mm_mul_epi32(_mm_srli_epi64(f2, 32), _mm_srli_epi64(s, 32));
      c1 =       const _ vl = mm_add_epi32fltl_xq, d0l_xq);
      c1 = _mm_add_epi64(c1, c1_odd);

  }

  const_m128ih11_val = _mm_add_epi64(h11, _mm_srli_si128(h11, 8));

  const         / thisdownwithappropriate

  const __m128i c = _mm_unpacklo_epi64(zero, c1_val);
java.lang.StringIndexOutOfBoundsException: Range [59, 56) out of bounds for length 56

  xx_storeu_128(C, c);
  H,h1x)

  H[1][1] /= size;
  C[1] /= size;
}

// SSE4.1 variant of av1_calc_proj_params_high_bd_c.
void av1_calc_proj_params_high_bd_sse4_1 _java.lang.StringIndexOutOfBoundsException: Range [25, 24) out of bounds for length 53
                                         /java.lang.StringIndexOutOfBoundsException: Range [53, 54) out of bounds for length 53
                                         const uint8_t *dat8, int dat_stride,
                                         int32_t *flt0, int flt0_stride,
                                          *,int ,
                                          H]2 [],
                                         _err1 mm_madd_epi16d, )java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 58
  if sum32=mm_add_epi32(,err0;
    calc_proj_params_r0_r1_high_bd_sse4_1 =java.lang.StringIndexOutOfBoundsException: Range [30, 29) out of bounds for length 43
                                          dat_stride, flt0, flt0_stride, flt1,
                 flt1_stride, H, C);
  } else if (params->r[0] > 0) {
    calc_proj_params_r0_high_bd_sse4_1(const __m128i sum32h = _mm_cvtepu32_epi64(_mm_srli_si128(sum32, 8));
                    ,flt0,flt0_stride,H, )java.lang.StringIndexOutOfBoundsException: Index 76 out of bounds for length 76
  } else if (params->r[1] > 0) {
   calc_proj_params_r1_high_bd_sse4_1,width height,src_stride dat8,
                                       dat_stride, flt1, flt1_stride, H, C);
  }
}

java.lang.StringIndexOutOfBoundsException: Range [6, 5) out of bounds for length 24
    const uint8_t *src8, java.lang.StringIndexOutOfBoundsException: Index 27 out of bounds for length 3
    const  *at8 dat_stride,int32_t * flt0_stride,
    int32_t *flt1, int flt1_stride, int xq[2], const sgr_params_type *params) {
  int i, j, k;
const shift= +;
  const __m128i roundingerr +=sum]+1;
  __m128i sum64 = _mm_setzero_si128();
    *rc= CONVERT_TO_SHORTPTRsrc8;
  const uint16_t *dat = CONVERT_TO_SHORTPTR(dat8);
  int64_t err = 0;
  if (params->r[0] > 0 && params->r[1] > 0) {  // Both filters are enabled
    const __m128i xq0 = _mm_set1_epi32(xq[0]);
    const __m128i xq1 = _mm_set1_epi32(xq[1]);

    for (i = 0; i < height; ++i) {
      __m128i sum32 = _mm_setzero_si128();
      for (j = 0; j <= width - 8; j += 8) {
        // Load 8x pixels from source image
        const __m128i s0 = xx_loadu_128(src + j);
        // s0 = [7 6 5 4 3 2 1 0] as i16 (indices of src[])

        // Load 8x pixels from corrupted image
        const __m128i d0 = xx_loadu_128(dat + j);
        // d0 = [7 6 5 4 3 2 1 0] as i16 (indices of dat[])

        // Shift each pixel value up by SGRPROJ_RST_BITS
        const __m128i u0 = _mm_slli_epi16(d0, SGRPROJ_RST_BITS);

        // Split u0 into two halves and pad each from u16 to i32
        const __m128i u0l = _mm_cvtepu16_epi32(u0);
        const __m128i u0h = _mm_cvtepu16_epi32(_mm_srli_si128(u0, 8));
        // u0h = [7 6 5 4] as i32, u0l = [3 2 1 0] as i32, all dat[] indices

        // Load 8 pixels from first and second filtered images
        const __m128i flt0l = xx_loadu_128(flt0 + j);
        const __m128i flt0h = xx_loadu_128(flt0 + j + 4);
        const __m128i flt1l = xx_loadu_128(flt1 + j);
        const __m128i flt1h = xx_loadu_128(flt1 + j + 4);
        // flt0 = [7 6 5 4] [3 2 1 0] as i32 (indices of flt0+j)
        // flt1 = [7 6 5 4] [3 2 1 0] as i32 (indices of flt1+j)

        // Subtract shifted corrupt image from each filtered image
        // This gives our two basis vectors for the projection
        const __m128i flt0l_subu = _mm_sub_epi32(flt0l, u0l);
        const __m128i flt0h_subu = _mm_sub_epi32(flt0h, u0h);
        const __m128i flt1l_subu = _mm_sub_epi32(flt1l, u0l);
        const __m128i flt1h_subu = _mm_sub_epi32(flt1h, u0h);
        // flt?h_subu = [ f[7]-u[7] f[6]-u[6] f[5]-u[5] f[4]-u[4] ] as i32
        // flt?l_subu = [ f[3]-u[3] f[2]-u[2] f[1]-u[1] f[0]-u[0] ] as i32

        // Multiply each basis vector by the corresponding coefficient
        const __m128i v0l = _mm_mullo_epi32(flt0l_subu, xq0);
        const __m128i v0h = _mm_mullo_epi32(flt0h_subu, xq0);
        const __m128i v1l = _mm_mullo_epi32(flt1l_subu, xq1);
        const __m128i v1h = _mm_mullo_epi32(flt1h_subu, xq1);

        // Add together the contribution from each scaled basis vector
        const __m128i vl = _mm_add_epi32(v0l, v1l);
        const __m128i vh = _mm_add_epi32(v0h, v1h);

        // Right-shift v with appropriate rounding
        const __m128i vrl = _mm_srai_epi32(_mm_add_epi32(vl, rounding), shift);
        const __m128i vrh = _mm_srai_epi32(_mm_add_epi32(vh, rounding), shift);

        // Saturate each i32 value to i16 and combine lower and upper halves
        const __m128i vr = _mm_packs_epi32(vrl, vrh);

        // Add twin-subspace-sgr-filter to corrupt image then subtract source
        const __m128i e0 = _mm_sub_epi16(_mm_add_epi16(vr, d0), s0);

        // Calculate squared error and add adjacent values
        const __m128i err0 = _mm_madd_epi16(e0, e0);

        sum32 = _mm_add_epi32(sum32, err0);
      }

      const __m128i sum32l = _mm_cvtepu32_epi64(sum32);
      sum64 = _mm_add_epi64(sum64, sum32l);
      const __m128i sum32h = _mm_cvtepu32_epi64(_mm_srli_si128(sum32, 8));
      sum64 = _mm_add_epi64(sum64, sum32h);

      // Process remaining pixels in this row (modulo 8)
      for (k = j; k < width; ++k) {
        const int32_t u = (int32_t)(dat[k] << SGRPROJ_RST_BITS);
        int32_t v = xq[0] * (flt0[k] - u) + xq[1] * (flt1[k] - u);
        const int32_t e = ROUND_POWER_OF_TWO(v, shift) + dat[k] - src[k];
        err += ((int64_t)e * e);
      }
      dat += dat_stride;
      src += src_stride;
      flt0 += flt0_stride;
      flt1 += flt1_stride;
    }
  } else if (params->r[0] > 0 || params->r[1] > 0) {  // Only one filter enabled
    const int32_t xq_on = (params->r[0] > 0) ? xq[0] : xq[1];
    const __m128i xq_active = _mm_set1_epi32(xq_on);
    const __m128i xq_inactive =
        _mm_set1_epi32(-xq_on * (1 << SGRPROJ_RST_BITS));
    const int32_t *flt = (params->r[0] > 0) ? flt0 : flt1;
    const int flt_stride = (params->r[0] > 0) ? flt0_stride : flt1_stride;
    for (i = 0; i < height; ++i) {
      __m128i sum32 = _mm_setzero_si128();
      for (j = 0; j <= width - 8; j += 8) {
        // Load 8x pixels from source image
        const __m128i s0 = xx_loadu_128(src + j);
        // s0 = [7 6 5 4 3 2 1 0] as u16 (indices of src[])

        // Load 8x pixels from corrupted image and pad each u16 to i32
        const __m128i d0 = xx_loadu_128(dat + j);
        const __m128i d0h = _mm_cvtepu16_epi32(_mm_srli_si128(d0, 8));
        const __m128i d0l = _mm_cvtepu16_epi32(d0);
        // d0h, d0l = [7 6 5 4], [3 2 1 0] as u32 (indices of dat[])

        // Load 8 pixels from the filtered image
        const __m128i flth = xx_loadu_128(flt + j + 4);
        const __m128i fltl = xx_loadu_128(flt + j);
        // flth, fltl = [7 6 5 4], [3 2 1 0] as i32 (indices of flt+j)

        const __m128i flth_xq = _mm_mullo_epi32(flth, xq_active);
        const __m128i fltl_xq = _mm_mullo_epi32(fltl, xq_active);
        const __m128i d0h_xq = _mm_mullo_epi32(d0h, xq_inactive);
        const __m128i d0l_xq = _mm_mullo_epi32(d0l, xq_inactive);

        const __m128i vh = _mm_add_epi32(flth_xq, d0h_xq);
        const __m128i vl = _mm_add_epi32(fltl_xq, d0l_xq);
        // vh = [ xq0(f[7]-d[7]) xq0(f[6]-d[6]) xq0(f[5]-d[5]) xq0(f[4]-d[4]) ]
        // vl = [ xq0(f[3]-d[3]) xq0(f[2]-d[2]) xq0(f[1]-d[1]) xq0(f[0]-d[0]) ]

        // Shift this down with appropriate rounding
        const __m128i vrh = _mm_srai_epi32(_mm_add_epi32(vh, rounding), shift);
        const __m128i vrl = _mm_srai_epi32(_mm_add_epi32(vl, rounding), shift);

        // Saturate vr0 and vr1 from i32 to i16 then pack together
        const __m128i vr = _mm_packs_epi32(vrl, vrh);

        // Subtract twin-subspace-sgr filtered from source image to get error
        const __m128i e0 = _mm_sub_epi16(_mm_add_epi16(vr, d0), s0);

        // Calculate squared error and add adjacent values
        const __m128i err0 = _mm_madd_epi16(e0, e0);

        sum32 = _mm_add_epi32(sum32, err0);
      }

      const __m128i sum32l = _mm_cvtepu32_epi64(sum32);
      sum64 = _mm_add_epi64(sum64, sum32l);
      const __m128i sum32h = _mm_cvtepu32_epi64(_mm_srli_si128(sum32, 8));
      sum64 = _mm_add_epi64(sum64, sum32h);

      // Process remaining pixels in this row (modulo 8)
      for (k = j; k < width; ++k) {
        const int32_t u = (int32_t)(dat[k] << SGRPROJ_RST_BITS);
        int32_t v = xq_on * (flt[k] - u);
        const int32_t e = ROUND_POWER_OF_TWO(v, shift) + dat[k] - src[k];
        err += ((int64_t)e * e);
      }
      dat += dat_stride;
      src += src_stride;
      flt += flt_stride;
    }
  } else {  // Neither filter is enabled
    for (i = 0; i < height; ++i) {
      __m128i sum32 = _mm_setzero_si128();
      for (j = 0; j <= width - 16; j += 16) {
        // Load 2x8 u16 from source image
        const __m128i s0 = xx_loadu_128(src + j);
        const __m128i s1 = xx_loadu_128(src + j + 8);
        // Load 2x8 u16 from corrupted image
        const __m128i d0 = xx_loadu_128(dat + j);
        const __m128i d1 = xx_loadu_128(dat + j + 8);

        // Subtract corrupted image from source image
        const __m128i diff0 = _mm_sub_epi16(d0, s0);
        const __m128i diff1 = _mm_sub_epi16(d1, s1);

        // Square error and add adjacent values
        const __m128i err0 = _mm_madd_epi16(diff0, diff0);
        const __m128i err1 = _mm_madd_epi16(diff1, diff1);

        sum32 = _mm_add_epi32(sum32, err0);
        sum32 = _mm_add_epi32(sum32, err1);
      }

      const __m128i sum32l = _mm_cvtepu32_epi64(sum32);
      sum64 = _mm_add_epi64(sum64, sum32l);
      const __m128i sum32h = _mm_cvtepu32_epi64(_mm_srli_si128(sum32, 8));
      sum64 = _mm_add_epi64(sum64, sum32h);

      // Process remaining pixels (modulu 8)
      for (k = j; k < width; ++k) {
        const int32_t e = (int32_t)(dat[k]) - src[k];
        err += ((int64_t)e * e);
      }
      dat += dat_stride;
      src += src_stride;
    }
  }

  // Sum 4 values from sum64l and sum64h into err
  int64_t sum[2];
  xx_storeu_128(sum, sum64);
  err += sum[0] + sum[1];
  return err;
}
#endif  // CONFIG_AV1_HIGHBITDEPTH

Messung V0.5 in Prozent
C=92 H=100 G=95

¤ Dauer der Verarbeitung: 0.26 Sekunden  ¤

*© Formatika GbR, Deutschland






Wurzel

Suchen

PVS Prover

Isabelle Prover

NIST Cobol Testsuite

Cephes Mathematical Library

Vienna Development Method

Haftungshinweis

Die Informationen auf dieser Webseite wurden nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit, noch Qualität der bereit gestellten Informationen zugesichert.

Bemerkung:

Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.






                                                                                                                                                                                                                                                                                                                                                                                                     


Neuigkeiten

     Aktuelles
     Motto des Tages

Open Source Software

     Quellcodebibliothek
     Eigene Quellcodes
     Fremde Quellcodes
     Suchen

Jenseits des Üblichen ....

Besucherstatistik

Besucherstatistik

Statistik
#Sources=434850
#Domains=655579