Created
October 28, 2016 01:13
-
-
Save hughperkins/a432442164d75be18b32bf780e480c8c to your computer and use it in GitHub Desktop.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| kernel void _ZN5Eigen8internal15EigenMetaKernelINS_15TensorEvaluatorIKNS_14TensorAssignOpINS_9TensorMapINS_6TensorIfLi2ELi0ElEELi0ENS_11MakePointerEEEKNS_17TensorReductionOpINS0_10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS4_INS5_IfLi4ELi0ElEELi0ES7_EEEEEENS_9GpuDeviceEEElEEvT_T0_(global struct Eigen__TensorEvaluator_5_nopointers* eval_nopointers, global float* eval_ptr0, long eval_ptr_offset0, global float* eval_ptr1, long eval_ptr_offset1, global float* eval_ptr2, long eval_ptr_offset2, long size) { | |
| float accum_0_i_i_i; | |
| float accum_1_i_i_i; | |
| float accum_1_i_i_i_unr; | |
| float accum_2_i_i_i_1_lcssa; | |
| float accum_2_i_i_i_lcssa; | |
| float accum_2_i_i_i_lcssa_unr; | |
| float accum_3_i_i_i; | |
| float accum_3_i_i_i_lcssa; | |
| long i_01_i; | |
| int j_01_i_i_i_i; | |
| int j_01_i_i_i_i_i; | |
| int j_01_i_i_i_i_i_unr; | |
| long v53; | |
| float v68; | |
| long v69; | |
| float v_0_i_i_i; | |
| float v_unr; | |
| long v_unr1; | |
| float v43; | |
| long v45; | |
| long v50; | |
| long v54; | |
| int v55; | |
| long v56; | |
| int v2; | |
| long v6; | |
| global float* v13; | |
| long v16; | |
| global float* v34; | |
| long v22; | |
| long v31; | |
| long v33; | |
| float v84; | |
| bool v85; | |
| long v57; | |
| float v62; | |
| bool v63; | |
| float accum_2_i_i_i_prol; | |
| float v64; | |
| long v71; | |
| float v76; | |
| bool v77; | |
| float v78; | |
| float accum_2_i_i_i_1; | |
| float v70; | |
| int v72; | |
| long v38; | |
| eval_ptr2 = (global float*)((global char *)eval_ptr2 + eval_ptr_offset2); | |
| eval_ptr1 = (global float*)((global char *)eval_ptr1 + eval_ptr_offset1); | |
| eval_ptr0 = (global float*)((global char *)eval_ptr0 + eval_ptr_offset0); | |
| struct Eigen__TensorEvaluator_5 eval[1]; | |
| eval[0].f0.f1.f0.f0[0] = eval_nopointers[0].f0.f1.f0.f0[0]; | |
| eval[0].f0.f1.f0.f0[1] = eval_nopointers[0].f0.f1.f0.f0[1]; | |
| eval[0].f1.f0.f0[0] = eval_nopointers[0].f1.f0.f0[0]; | |
| eval[0].f1.f0.f0[1] = eval_nopointers[0].f1.f0.f0[1]; | |
| eval[0].f1.f0.f0[2] = eval_nopointers[0].f1.f0.f0[2]; | |
| eval[0].f1.f0.f0[3] = eval_nopointers[0].f1.f0.f0[3]; | |
| eval[0].f1.f1.f0.f0[0] = eval_nopointers[0].f1.f1.f0.f0[0]; | |
| eval[0].f1.f1.f0.f0[1] = eval_nopointers[0].f1.f1.f0.f0[1]; | |
| eval[0].f1.f2.f0[0] = eval_nopointers[0].f1.f2.f0[0]; | |
| eval[0].f1.f2.f0[1] = eval_nopointers[0].f1.f2.f0[1]; | |
| eval[0].f1.f3.f0[0] = eval_nopointers[0].f1.f3.f0[0]; | |
| eval[0].f1.f3.f0[1] = eval_nopointers[0].f1.f3.f0[1]; | |
| eval[0].f1.f4.f0[0] = eval_nopointers[0].f1.f4.f0[0]; | |
| eval[0].f1.f4.f0[1] = eval_nopointers[0].f1.f4.f0[1]; | |
| eval[0].f1.f5.f0[0] = eval_nopointers[0].f1.f5.f0[0]; | |
| eval[0].f1.f5.f0[1] = eval_nopointers[0].f1.f5.f0[1]; | |
| eval[0].f1.f6.f1.f0.f0[0] = eval_nopointers[0].f1.f6.f1.f0.f0[0]; | |
| eval[0].f1.f6.f1.f0.f0[1] = eval_nopointers[0].f1.f6.f1.f0.f0[1]; | |
| eval[0].f1.f6.f1.f0.f0[2] = eval_nopointers[0].f1.f6.f1.f0.f0[2]; | |
| eval[0].f1.f6.f1.f0.f0[3] = eval_nopointers[0].f1.f6.f1.f0.f0[3]; | |
| eval[0].f1.f7.f0 = eval_nopointers[0].f1.f7.f0; | |
| eval[0].f0.f0 = eval_ptr0; | |
| eval[0].f1.f6.f0 = eval_ptr1; | |
| eval[0].f1.f8 = eval_ptr2; | |
| label0:; | |
| v2 = get_local_size(0); | |
| v6 = (v2 * get_group_id(0)) + get_local_id(0); | |
| if(!(v6 < size)) { | |
| goto _ZN5Eigen8internal19EigenMetaKernelEvalINS_15TensorEvaluatorIKNS_14TensorAssignOpINS_9TensorMapINS_6TensorIfLi2ELi0ElEELi0ENS_11MakePointerEEEKNS_17TensorReductionOpINS0_10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS4_INS5_IfLi4ELi0ElEELi0ES7_EEEEEENS_9GpuDeviceEEElLb0EE3runERSN_lll_exit; | |
| } | |
| v_lr_ph_i:; | |
| v13 = (&eval[0].f1.f8)[0]; | |
| v16 = (&eval[0].f1.f2.f0[1])[0]; | |
| v22 = (&eval[0].f1.f5.f0[1])[0]; | |
| v31 = (&eval[0].f1.f5.f0[0])[0]; | |
| v33 = (&eval[0].f1.f4.f0[0])[0]; | |
| v34 = (&eval[0].f1.f6.f0)[0]; | |
| i_01_i = v6; | |
| v37:; | |
| if (v13 == 0) { | |
| goto v41; | |
| } | |
| v40:; | |
| v43 = (&v13[i_01_i])[0]; | |
| v_0_i_i_i = v43; | |
| goto _ZN5Eigen15TensorEvaluatorIKNS_14TensorAssignOpINS_9TensorMapINS_6TensorIfLi2ELi0ElEELi0ENS_11MakePointerEEEKNS_17TensorReductionOpINS_8internal10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS2_INS3_IfLi4ELi0ElEELi0ES5_EEEEEENS_9GpuDeviceEE10evalScalarEl_exit_i; | |
| v41:; | |
| v45 = i_01_i / v16; | |
| v50 = ((i_01_i - (v45 * v16)) * ((&eval[0].f1.f3.f0[0])[0])) + (v45 * ((&eval[0].f1.f3.f0[1])[0])); | |
| if(!(v22 > 0)) { | |
| v_0_i_i_i = -INFINITY; | |
| goto _ZN5Eigen15TensorEvaluatorIKNS_14TensorAssignOpINS_9TensorMapINS_6TensorIfLi2ELi0ElEELi0ENS_11MakePointerEEEKNS_17TensorReductionOpINS_8internal10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS2_INS3_IfLi4ELi0ElEELi0ES5_EEEEEENS_9GpuDeviceEE10evalScalarEl_exit_i; | |
| } | |
| v_lr_ph_i_i_i_i_preheader:; | |
| accum_0_i_i_i = -INFINITY; | |
| v53 = 0; | |
| j_01_i_i_i_i = 0; | |
| v_lr_ph_i_i_i_i:; | |
| v56 = v53 * ((&eval[0].f1.f4.f0[1])[0]); | |
| v57 = v50 + v56; | |
| if(!(v31 > 0)) { | |
| accum_3_i_i_i = accum_0_i_i_i; | |
| goto _ZN5Eigen8internal17GenericDimReducerILi0ENS_15TensorEvaluatorIKNS_17TensorReductionOpINS0_10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS_9TensorMapINS_6TensorIfLi4ELi0ElEELi0ENS_11MakePointerEEEEENS_9GpuDeviceEEES5_E6reduceERKSI_lRS5_Pf_exit_i_i_i_i; | |
| } | |
| _ZNK5Eigen8internal10MaxReducerIfE6reduceEfPf_exit_i_i_i_i_i_preheader:; | |
| if ((v31 & 1) == 0) { | |
| accum_1_i_i_i_unr = accum_0_i_i_i; | |
| v_unr = accum_0_i_i_i; | |
| v_unr1 = 0; | |
| j_01_i_i_i_i_i_unr = 0; | |
| goto _ZNK5Eigen8internal10MaxReducerIfE6reduceEfPf_exit_i_i_i_i_i_preheader_split; | |
| } | |
| _ZNK5Eigen8internal10MaxReducerIfE6reduceEfPf_exit_i_i_i_i_i_prol:; | |
| v62 = (&v34[v50 + v56])[0]; | |
| v63 = accum_0_i_i_i < v62; | |
| accum_2_i_i_i_prol = v63 ? v62 : accum_0_i_i_i; | |
| v64 = v63 ? v62 : accum_0_i_i_i; | |
| accum_1_i_i_i_unr = accum_2_i_i_i_prol; | |
| v_unr = v64; | |
| v_unr1 = 1; | |
| j_01_i_i_i_i_i_unr = 1; | |
| accum_2_i_i_i_lcssa_unr = accum_2_i_i_i_prol; | |
| _ZNK5Eigen8internal10MaxReducerIfE6reduceEfPf_exit_i_i_i_i_i_preheader_split:; | |
| if (v31 == 1) { | |
| accum_2_i_i_i_lcssa = accum_2_i_i_i_lcssa_unr; | |
| goto _ZN5Eigen8internal17GenericDimReducerILi0ENS_15TensorEvaluatorIKNS_17TensorReductionOpINS0_10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS_9TensorMapINS_6TensorIfLi4ELi0ElEELi0ENS_11MakePointerEEEEENS_9GpuDeviceEEES5_E6reduceERKSI_lRS5_Pf_exit_i_i_i_i_loopexit; | |
| } | |
| _ZNK5Eigen8internal10MaxReducerIfE6reduceEfPf_exit_i_i_i_i_i_preheader_split_split:; | |
| accum_1_i_i_i = accum_1_i_i_i_unr; | |
| v68 = v_unr; | |
| v69 = v_unr1; | |
| j_01_i_i_i_i_i = j_01_i_i_i_i_i_unr; | |
| _ZNK5Eigen8internal10MaxReducerIfE6reduceEfPf_exit_i_i_i_i_i:; | |
| v76 = (&v34[v57 + (v69 * v33)])[0]; | |
| v77 = v68 < v76; | |
| v78 = v77 ? v76 : v68; | |
| v84 = (&v34[v57 + ((j_01_i_i_i_i_i + 1) * v33)])[0]; | |
| v85 = v78 < v84; | |
| accum_2_i_i_i_1 = v85 ? v84 : (v77 ? v76 : accum_1_i_i_i); | |
| v70 = v85 ? v84 : v78; | |
| v72 = j_01_i_i_i_i_i + 2; | |
| v71 = v72; | |
| if (v71 < v31) { | |
| accum_1_i_i_i = accum_2_i_i_i_1; | |
| v68 = v70; | |
| v69 = v71; | |
| j_01_i_i_i_i_i = v72; | |
| goto _ZNK5Eigen8internal10MaxReducerIfE6reduceEfPf_exit_i_i_i_i_i; | |
| } else { | |
| accum_2_i_i_i_1_lcssa = accum_2_i_i_i_1; | |
| } | |
| _ZN5Eigen8internal17GenericDimReducerILi0ENS_15TensorEvaluatorIKNS_17TensorReductionOpINS0_10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS_9TensorMapINS_6TensorIfLi4ELi0ElEELi0ENS_11MakePointerEEEEENS_9GpuDeviceEEES5_E6reduceERKSI_lRS5_Pf_exit_i_i_i_i_loopexit_unr_lcssa:; | |
| accum_2_i_i_i_lcssa = accum_2_i_i_i_1_lcssa; | |
| _ZN5Eigen8internal17GenericDimReducerILi0ENS_15TensorEvaluatorIKNS_17TensorReductionOpINS0_10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS_9TensorMapINS_6TensorIfLi4ELi0ElEELi0ENS_11MakePointerEEEEENS_9GpuDeviceEEES5_E6reduceERKSI_lRS5_Pf_exit_i_i_i_i_loopexit:; | |
| accum_3_i_i_i = accum_2_i_i_i_lcssa; | |
| _ZN5Eigen8internal17GenericDimReducerILi0ENS_15TensorEvaluatorIKNS_17TensorReductionOpINS0_10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS_9TensorMapINS_6TensorIfLi4ELi0ElEELi0ENS_11MakePointerEEEEENS_9GpuDeviceEEES5_E6reduceERKSI_lRS5_Pf_exit_i_i_i_i:; | |
| v55 = j_01_i_i_i_i + 1; | |
| v54 = v55; | |
| if (v54 < v22) { | |
| accum_0_i_i_i = accum_3_i_i_i; | |
| v53 = v54; | |
| j_01_i_i_i_i = v55; | |
| goto v_lr_ph_i_i_i_i; | |
| } else { | |
| accum_3_i_i_i_lcssa = accum_3_i_i_i; | |
| } | |
| _ZN5Eigen15TensorEvaluatorIKNS_14TensorAssignOpINS_9TensorMapINS_6TensorIfLi2ELi0ElEELi0ENS_11MakePointerEEEKNS_17TensorReductionOpINS_8internal10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS2_INS3_IfLi4ELi0ElEELi0ES5_EEEEEENS_9GpuDeviceEE10evalScalarEl_exit_i_loopexit:; | |
| v_0_i_i_i = accum_3_i_i_i_lcssa; | |
| _ZN5Eigen15TensorEvaluatorIKNS_14TensorAssignOpINS_9TensorMapINS_6TensorIfLi2ELi0ElEELi0ENS_11MakePointerEEEKNS_17TensorReductionOpINS_8internal10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS2_INS3_IfLi4ELi0ElEELi0ES5_EEEEEENS_9GpuDeviceEE10evalScalarEl_exit_i:; | |
| (&((&eval[0].f0.f0)[0])[i_01_i])[0] = v_0_i_i_i; | |
| v38 = i_01_i + (get_num_groups(0) * v2); | |
| if (v38 < size) { | |
| i_01_i = v38; | |
| goto v37; | |
| } | |
| _ZN5Eigen8internal19EigenMetaKernelEvalINS_15TensorEvaluatorIKNS_14TensorAssignOpINS_9TensorMapINS_6TensorIfLi2ELi0ElEELi0ENS_11MakePointerEEEKNS_17TensorReductionOpINS0_10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS4_INS5_IfLi4ELi0ElEELi0ES7_EEEEEENS_9GpuDeviceEEElLb0EE3runERSN_lll_exit_loopexit:; | |
| _ZN5Eigen8internal19EigenMetaKernelEvalINS_15TensorEvaluatorIKNS_14TensorAssignOpINS_9TensorMapINS_6TensorIfLi2ELi0ElEELi0ENS_11MakePointerEEEKNS_17TensorReductionOpINS0_10MaxReducerIfEEKNS_5arrayIlLm2EEEKNS4_INS5_IfLi4ELi0ElEELi0ES7_EEEEEENS_9GpuDeviceEEElLb0EE3runERSN_lll_exit:; | |
| return; | |
| } |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment