10d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar///***************************************************************************** 20d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 30d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* Copyright (C) 2012 Ittiam Systems Pvt Ltd, Bangalore 40d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 50d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* Licensed under the Apache License, Version 2.0 (the "License"); 60d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* you may not use this file except in compliance with the License. 70d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* You may obtain a copy of the License at: 80d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 90d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* http://www.apache.org/licenses/LICENSE-2.0 100d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 110d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* Unless required by applicable law or agreed to in writing, software 120d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* distributed under the License is distributed on an "AS IS" BASIS, 130d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. 140d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* See the License for the specific language governing permissions and 150d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* limitations under the License. 160d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 170d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//*****************************************************************************/ 180d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar///** 190d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//******************************************************************************* 200d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @file 210d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* ihevc_intra_pred_filters_planar.s 220d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 230d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @brief 240d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* contains function definitions for inter prediction interpolation. 250d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* functions are coded using neon intrinsics and can be compiled using 260d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 270d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* rvct 280d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 290d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @author 300d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* akshaya mukund 310d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 320d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @par list of functions: 330d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 340d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 350d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @remarks 360d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* none 370d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 380d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//******************************************************************************* 390d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//*/ 400d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar///** 410d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//******************************************************************************* 420d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 430d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @brief 440d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* luma intraprediction filter for planar input 450d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 460d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @par description: 470d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 480d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @param[in] pu1_ref 490d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* uword8 pointer to the source 500d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 510d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @param[out] pu1_dst 520d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* uword8 pointer to the destination 530d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 540d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @param[in] src_strd 550d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* integer source stride 560d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 570d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @param[in] dst_strd 580d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* integer destination stride 590d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 600d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @param[in] pi1_coeff 610d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* word8 pointer to the planar coefficients 620d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 630d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @param[in] nt 640d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* size of tranform block 650d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 660d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @param[in] mode 670d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* type of filtering 680d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 690d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @returns 700d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 710d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* @remarks 720d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* none 730d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//* 740d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//******************************************************************************* 750d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//*/ 760d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 770d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//void ihevc_intra_pred_luma_planar(uword8* pu1_ref, 780d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// word32 src_strd, 790d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// uword8* pu1_dst, 800d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// word32 dst_strd, 810d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// word32 nt, 820d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// word32 mode, 830d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// word32 pi1_coeff) 840d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//**************variables vs registers***************************************** 850d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//x0 => *pu1_ref 860d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//x1 => src_strd 870d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//x2 => *pu1_dst 880d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//x3 => dst_strd 890d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 900d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar//stack contents from #40 910d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// nt 920d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// mode 930d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// pi1_coeff 940d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 950d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar.text 960d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar.align 4 970d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar.include "ihevc_neon_macros.s" 980d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 990d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1000d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar.globl ihevc_intra_pred_chroma_planar_av8 1010d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar.extern gau1_ihevc_planar_factor 1020d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1030d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1040d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar.type ihevc_intra_pred_chroma_planar_av8, %function 1050d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1060d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakarihevc_intra_pred_chroma_planar_av8: 1070d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1080d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar // stmfd sp!, {x4-x12, x14} //stack stores the values of the arguments 1099cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy 1109cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy stp d10,d11,[sp,#-16]! 1119cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy stp d12,d13,[sp,#-16]! 1129cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy stp d8,d14,[sp,#-16]! // Storing d14 using { sub sp,sp,#8; str d14,[sp] } is giving bus error. 1139cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy // d8 is used as dummy register and stored along with d14 using stp. d8 is not used in the function. 1140d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar stp x19, x20,[sp,#-16]! 1150d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1160d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar adrp x11, :got:gau1_ihevc_planar_factor //loads table of coeffs 1170d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ldr x11, [x11, #:got_lo12:gau1_ihevc_planar_factor] 1180d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1190d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar clz w5,w4 1200d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sub x20, x5, #32 1210d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar neg x5, x20 1220d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v14.8h,w5 1230d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar neg v14.8h, v14.8h //shr value (so vneg) 1240d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v2.8b,w4 //nt 1250d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v16.8h,w4 //nt 1260d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1270d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sub x6, x4, #1 //nt-1 1280d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add x6, x0,x6,lsl #1 //2*(nt-1) 1290d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ldr w7, [x6] 1300d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sxtw x7,w7 1310d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v0.4h,w7 //src[nt-1] 1320d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1330d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add x6, x4, x4,lsl #1 //3nt 1340d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add x6, x6, #1 //3nt + 1 1350d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar lsl x6,x6,#1 //2*(3nt + 1) 1360d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1370d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add x6, x6, x0 1380d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ldr w7, [x6] 1390d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sxtw x7,w7 1400d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v1.4h,w7 //src[3nt+1] 1410d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1420d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1430d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add x6, x4, x4 //2nt 1440d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add x14, x6, #1 //2nt+1 1450d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar lsl x14,x14,#1 //2*(2nt+1) 1460d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sub x6, x6, #1 //2nt-1 1470d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar lsl x6,x6,#1 //2*(2nt-1) 1480d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add x6, x6, x0 //&src[2nt-1] 1490d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add x14, x14, x0 //&src[2nt+1] 1500d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1510d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar mov x8, #1 //row+1 (row is first 0) 1520d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sub x9, x4, x8 //nt-1-row (row is first 0) 1530d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1540d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v5.8b,w8 //row + 1 1550d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v6.8b,w9 //nt - 1 - row 1560d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar mov v7.8b, v5.8b //mov #1 to d7 to used for inc for row+1 and dec for nt-1-row 1570d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1580d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add x12, x11, #1 //coeffs (to be reloaded after every row) 1590d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar mov x1, x4 //nt (row counter) (dec after every row) 1600d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar mov x5, x2 //dst (to be reloaded after every row and inc by dst_strd) 1610d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar mov x10, #8 //increment for the coeffs 1620d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar mov x0, x14 //&src[2nt+1] (to be reloaded after every row) 1630d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1640d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar cmp x4, #4 1650d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar beq tf_sz_4 1660d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1670d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1680d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1690d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar mov x10,x6 1700d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakartf_sz_8_16: 1710d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ld1 {v10.8b, v11.8b}, [x14],#16 //load src[2nt+1+col] 1729cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy ld1 {v17.8b},[x12],#8 1739cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy mov v25.8b, v17.8b 1749cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy zip1 v29.8b, v17.8b, v25.8b 1759cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy zip2 v25.8b, v17.8b, v25.8b 1769cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy mov v17.d[0], v29.d[0] 1779cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy sub v30.8b, v2.8b , v17.8b //[nt-1-col] 1789cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy sub v31.8b, v2.8b , v25.8b 1790d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1800d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1810d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1820d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1830d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakarloop_sz_8_16: 1840d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1850d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ldr w7, [x6], #-2 //src[2nt-1-row] (dec to take into account row) 1860d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sxtw x7,w7 1870d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umull v12.8h, v5.8b, v0.8b //(row+1) * src[nt-1] 1880d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ldr w11, [x6], #-2 //src[2nt-1-row] (dec to take into account row) 1890d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sxtw x11,w11 1900d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v12.8h, v6.8b, v10.8b //(nt-1-row) * src[2nt+1+col] 1910d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v4.4h,w7 //src[2nt-1-row] 1929cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy umlal v12.8h, v17.8b, v1.8b //(col+1) * src[3nt+1] 1930d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v3.4h,w11 //src[2nt-1-row] 1940d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v12.8h, v30.8b, v4.8b //(nt-1-col) * src[2nt-1-row] 1950d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1960d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1970d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 1980d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umull v28.8h, v5.8b, v0.8b 1990d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ldr w7, [x6], #-2 //src[2nt-1-row] (dec to take into account row) 2000d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sxtw x7,w7 2010d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v28.8h, v6.8b, v11.8b 2020d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v18.8b, v5.8b , v7.8b //row++ [(row+1)++]c 2030d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2040d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2050d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v28.8h, v31.8b, v4.8b 2060d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sub v19.8b, v6.8b , v7.8b //[nt-1-row]-- 2079cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy umlal v28.8h, v25.8b, v1.8b 2080d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v4.4h,w7 //src[2nt-1-row] 2090d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2100d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umull v26.8h, v18.8b, v0.8b //(row+1) * src[nt-1] 2110d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v12.8h, v12.8h , v16.8h //add (nt) 2120d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v26.8h, v19.8b, v10.8b //(nt-1-row) * src[2nt+1+col] 2130d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sshl v12.8h, v12.8h, v14.8h //shr 2149cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy umlal v26.8h, v17.8b, v1.8b //(col+1) * src[3nt+1] 2150d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v28.8h, v28.8h , v16.8h 2160d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v26.8h, v30.8b, v3.8b //(nt-1-col) * src[2nt-1-row] 2170d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sshl v28.8h, v28.8h, v14.8h 2180d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2190d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2200d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2210d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2220d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2230d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umull v24.8h, v18.8b, v0.8b 2240d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v5.8b, v18.8b , v7.8b //row++ [(row+1)++] 2250d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v24.8h, v19.8b, v11.8b 2260d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sub v6.8b, v19.8b , v7.8b //[nt-1-row]-- 2279cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy umlal v24.8h, v25.8b, v1.8b 2280d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar xtn v12.8b, v12.8h 2290d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v24.8h, v31.8b, v3.8b 2300d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar xtn v13.8b, v28.8h 2310d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2320d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2330d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2340d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2350d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v26.8h, v26.8h , v16.8h //add (nt) 2360d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umull v22.8h, v5.8b, v0.8b //(row+1) * src[nt-1] 2370d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sshl v26.8h, v26.8h, v14.8h //shr 2380d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v22.8h, v6.8b, v10.8b //(nt-1-row) * src[2nt+1+col] 2390d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar st1 {v12.2s, v13.2s}, [x2], x3 2409cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy umlal v22.8h, v17.8b, v1.8b //(col+1) * src[3nt+1] 2410d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v24.8h, v24.8h , v16.8h 2420d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v22.8h, v30.8b, v4.8b //(nt-1-col) * src[2nt-1-row] 2430d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sshl v24.8h, v24.8h, v14.8h 2440d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2450d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umull v20.8h, v5.8b, v0.8b 2460d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v18.8b, v5.8b , v7.8b //row++ [(row+1)++]c 2470d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v20.8h, v6.8b, v11.8b 2480d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sub v19.8b, v6.8b , v7.8b //[nt-1-row]-- 2490d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v20.8h, v31.8b, v4.8b 2500d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2510d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ldr w11, [x6], #-2 //src[2nt-1-row] (dec to take into account row) 2520d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sxtw x11,w11 2539cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy umlal v20.8h, v25.8b, v1.8b 2540d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v3.4h,w11 //src[2nt-1-row] 2550d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v22.8h, v22.8h , v16.8h //add (nt) 2560d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2570d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umull v12.8h, v18.8b, v0.8b //(row+1) * src[nt-1] 2580d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar xtn v26.8b, v26.8h 2590d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v12.8h, v19.8b, v10.8b //(nt-1-row) * src[2nt+1+col] 2600d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar xtn v27.8b, v24.8h 2610d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2629cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy umlal v12.8h, v17.8b, v1.8b //(col+1) * src[3nt+1] 2630d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sshl v22.8h, v22.8h, v14.8h //shr 2640d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2650d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v12.8h, v30.8b, v3.8b //(nt-1-col) * src[2nt-1-row] 2660d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v20.8h, v20.8h , v16.8h 2670d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2680d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umull v28.8h, v18.8b, v0.8b 2690d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar st1 {v26.2s, v27.2s}, [x2], x3 2700d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2710d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v28.8h, v19.8b, v11.8b 2720d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v5.8b, v18.8b , v7.8b //row++ [(row+1)++] 2730d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2740d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sub v6.8b, v19.8b , v7.8b //[nt-1-row]-- 2759cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy umlal v28.8h, v25.8b, v1.8b 2760d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2770d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v28.8h, v31.8b, v3.8b 2780d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sshl v20.8h, v20.8h, v14.8h 2790d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2800d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2810d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v12.8h, v12.8h , v16.8h //add (nt) 2820d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar xtn v22.8b, v22.8h 2830d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2840d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2850d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v28.8h, v28.8h , v16.8h 2860d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar xtn v23.8b, v20.8h 2870d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2880d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2890d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sshl v12.8h, v12.8h, v14.8h //shr 2900d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar st1 {v22.2s, v23.2s}, [x2], x3 2910d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sshl v28.8h, v28.8h, v14.8h 2920d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2930d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2940d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2950d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2960d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 2970d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar xtn v20.8b, v12.8h 2980d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar xtn v21.8b, v28.8h 2990d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3000d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar st1 {v20.2s, v21.2s}, [x2], x3 3010d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3020d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3030d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar subs x1, x1, #4 3040d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3050d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar bne loop_sz_8_16 3060d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3070d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3080d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3090d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3100d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar cmp x4,#16 3110d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3120d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar bne end_loop 3130d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3140d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3150d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sub x4, x4,#16 3160d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v5.8b,w8 //row + 1 3170d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v6.8b,w9 //nt - 1 - row 3180d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar mov v7.8b, v5.8b //mov #1 to d7 to used for inc for row+1 and dec for nt-1-row 3190d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3200d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar mov x6,x10 3210d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar mov x1,#16 3220d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sub x2,x2,x3,lsl #4 3230d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add x2,x2,#16 3240d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3250d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ld1 {v10.8b, v11.8b}, [x14],#16 //load src[2nt+1+col] 3269cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy ld1 {v17.8b},[x12],#8 3279cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy mov v25.8b, v17.8b 3289cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy zip1 v29.8b, v17.8b, v25.8b 3299cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy zip2 v25.8b, v17.8b, v25.8b 3309cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy mov v17.d[0], v29.d[0] 3319cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy sub v30.8b, v2.8b , v17.8b //[nt-1-col] 3329cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy sub v31.8b, v2.8b , v25.8b 3330d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3340d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar beq loop_sz_8_16 3350d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3360d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3370d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3380d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakartf_sz_4: 3390d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ld1 {v10.8b},[x14] //load src[2nt+1+col] 3409cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy ld1 {v17.8b},[x12], x10 //load 8 coeffs [col+1] 3419cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy mov v25.8b, v17.8b 3429cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy zip1 v29.8b, v17.8b, v25.8b 3439cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy zip2 v25.8b, v17.8b, v25.8b 3449cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy mov v17.d[0], v29.d[0] 3450d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakarloop_sz_4: 3460d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar //mov x10, #4 @reduce inc to #4 for 4x4 3470d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ldr w7, [x6], #-2 //src[2nt-1-row] (dec to take into account row) 3480d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sxtw x7,w7 3490d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar dup v4.4h,w7 //src[2nt-1-row] 3500d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3519cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy sub v25.8b, v2.8b , v17.8b //[nt-1-col] 3520d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3530d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umull v12.8h, v5.8b, v0.8b //(row+1) * src[nt-1] 3540d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar umlal v12.8h, v6.8b, v10.8b //(nt-1-row) * src[2nt+1+col] 3559cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy umlal v12.8h, v17.8b, v1.8b //(col+1) * src[3nt+1] 3569cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy umlal v12.8h, v25.8b, v4.8b //(nt-1-col) * src[2nt-1-row] 3570d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// vadd.i16 q6, q6, q8 @add (nt) 3580d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// vshl.s16 q6, q6, q7 @shr 3590d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar// vmovn.i16 d12, q6 3600d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar rshrn v12.8b, v12.8h,#3 3610d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3620d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar st1 {v12.2s},[x2], x3 3630d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3640d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar add v5.8b, v5.8b , v7.8b //row++ [(row+1)++] 3650d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar sub v6.8b, v6.8b , v7.8b //[nt-1-row]-- 3660d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar subs x1, x1, #1 3670d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3680d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar bne loop_sz_4 3690d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3700d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakarend_loop: 3719cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy // ldmfd sp!,{x4-x12,x15} //reload the registers from sp 3720d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ldp x19, x20,[sp],#16 3739cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy ldp d8,d14,[sp],#16 // Loading d14 using { ldr d14,[sp]; add sp,sp,#8 } is giving bus error. 3749cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy // d8 is used as dummy register and loaded along with d14 using ldp. d8 is not used in the function. 3759cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy ldp d12,d13,[sp],#16 3769cbd70a2930875be59d7df68136ac9a1a949a13dNaveen Kumar Ponnusamy ldp d10,d11,[sp],#16 3770d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar ret 3780d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3790d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3800d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3810d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3820d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3830d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 3840d8951cef4b1a1dbf4ff5ba3e8796cf1d4503098Harish Mahendrakar 385