hschumann2/TempleOS-Source-Code
0847
1 2/*The moral of this story is simple3inst level optimizations4don't matter much on a modern Intel CPU5because they convert complex insts6to a stream of RISC insts.7 8I learned this the hard way when I thought9I was greatly improving my compiler by10cutting code by a third. No significant11speed-up. Depressing.12*/13 14#define SAMPLES (8*10000000+1)15 16asm {17 18LIMIT:: DU64 SAMPLES; //Memory reference should be bad, right?19 20_BADLY_UNOPTIMIZED::21 MOV RAX,022 MOV RCX,123@@05: MOV RDX,RCX24 INC RCX //if no dependencies, Free!25 ADD RAX,RDX26 MOV RDX,LIMIT-16 //added 16 displacement to make it worse27 CMP RCX,U64 16[RDX]28 JB @@0529 RET30 31_WELL_OPTIMIZED1::32 XOR RAX,RAX33 MOV RCX,SAMPLES-134@@05: ADD RAX,RCX35 DEC RCX36 JNZ @@0537 RET38 39_WELL_OPTIMIZED2:: //Unrolled40 XOR RAX,RAX41 MOV RCX,SAMPLES-142@@05: ADD RAX,RCX43 DEC RCX44 ADD RAX,RCX45 DEC RCX46 ADD RAX,RCX47 DEC RCX48 ADD RAX,RCX49 DEC RCX50 ADD RAX,RCX51 DEC RCX52 ADD RAX,RCX53 DEC RCX54 ADD RAX,RCX55 DEC RCX56 ADD RAX,RCX57 DEC RCX58 JNZ @@0559 RET60 61_WELL_OPTIMIZED3::62 XOR RAX,RAX63 MOV RCX,SAMPLES-164@@05: ADD RAX,RCX65 LOOP @@05 //Inst has slow speed, but saves code size.66 RET67}68 69_extern _BADLY_UNOPTIMIZED I64 Loop1();70_extern _WELL_OPTIMIZED1 I64 Loop2();71_extern _WELL_OPTIMIZED2 I64 Loop3();72_extern _WELL_OPTIMIZED3 I64 Loop4();73 74I64 i;75F64 t0;76 77CPURep;78 79"Bad Code\n";80t0=tS;81i=Loop1;82"Res:%d Time:%9.6f\n",i,tS-t0;83 84"Good Code #1\n";85t0=tS;86i=Loop2;87"Res:%d Time:%9.6f\n",i,tS-t0;88 89"Good Code #2\n";90t0=tS;91i=Loop3;92"Res:%d Time:%9.6f\n",i,tS-t0;93 94"Good Code #3\n";95t0=tS;96i=Loop4;97"Res:%d Time:%9.6f\n",i,tS-t0;98 99/* Program Output1008 Cores 2.660GHz101Bad Code102Res:3200000040000000 Time: 0.069966103Good Code #1104Res:3200000040000000 Time: 0.062567105Good Code #2106Res:3200000040000000 Time: 0.062907107Good Code #3108Res:3200000040000000 Time: 0.156359109*/110 