Apple Microarchitecture Research by Dougall Johnson

M1/A14 P-core (Firestorm): Overview | Base Instructions | SIMD and FP Instructions
M1/A14 E-core (Icestorm):  Overview | Base Instructions | SIMD and FP Instructions

FCCMPE (scalar, S)

Test 1: uops

Code:

  fccmpe s0, s1, #0, lt
  mov x0, 1
  mov x1, 2
  mov x2, 3
  mov x3, 4

(no loop instructions)

1000 unrolls and 1 iteration

Retires: 1.000

Issues: 1.000

Integer unit issues: 0.001

Load/store unit issues: 0.000

SIMD/FP unit issues: 1.000

retire uop (01)cycle (02)schedule uop (52)schedule int uop (53)schedule simd uop (54)dispatch simd uop (57)ldst uops in schedulers (5b)dispatch uop (78)map simd uop (7e)map simd uop inputs (81)? int output thing (e9)
100420331001110001000247691000100030001
100420331001110001000247691000100030001
100420331001110001000247691000100030001
100420331001110001000247691000100030001
100420331001110001000247691000100030001
100420331001110001000247691000100030001
100420331001110001000247691000100030001
100420331001110001000247691000100030001
100420331001110001000247691000100030001
100420331001110001000247691000100030001

Test 2: Latency 3->1

Chain cycles: 2

Code:

  fccmpe s0, s1, #0, lt
  fcsel d0, d2, d3, eq
  mov x0, 1
  mov x1, 2
  mov x2, 3
  mov x3, 4

(fused SUBS/B.cc loop)

100 unrolls and 100 iterations

Result (median cycles for code, minus 2 chain cycles): 2.0033

retire uop (01)cycle (02)schedule uop (52)schedule int uop (53)schedule simd uop (54)dispatch int uop (56)dispatch simd uop (57)int uops in schedulers (59)ldst uops in schedulers (5b)dispatch uop (78)map int uop (7c)map simd uop (7e)map int uop inputs (7f)map simd uop inputs (81)? int output thing (e9)? simd retires (ee)? int retires (ef)
20204400332010110120000100200003001019248201002002000620060012110000100
20204400332010110120000100200003001019248201002002000420060012110000100
20204400332010110120000100200003001019248201002002000420060012110000100
20204400332010110120000100200003001019248201002002000420060012110000100
20204400332010110120000100200003001019248201002002000420060012110000100
20204400332010110120000100200003001019248201002002000420060012110000100
20204400332010110120000100200003001019248201002002000420060012110000100
20204400332010110120000100200003001019248201002002000420060012110000100
20204400332010110120000100200003001019248201002002000420060012110000100
20204400332010110120000100200003001019248201002002000420060012110000100

1000 unrolls and 10 iterations

Result (median cycles for code, minus 2 chain cycles): 2.0033

retire uop (01)cycle (02)schedule uop (52)schedule int uop (53)schedule simd uop (54)dispatch int uop (56)dispatch simd uop (57)int uops in schedulers (59)ldst uops in schedulers (5b)dispatch uop (78)map int uop (7c)map simd uop (7e)map int uop inputs (7f)map simd uop inputs (81)? int output thing (e9)? simd retires (ee)? int retires (ef)
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20025400662001911200081020034301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010

Test 3: Latency 3->2

Chain cycles: 2

Code:

  fccmpe s0, s1, #0, lt
  fcsel d1, d2, d3, eq
  mov x0, 1
  mov x1, 2
  mov x2, 3
  mov x3, 4

(fused SUBS/B.cc loop)

100 unrolls and 100 iterations

Result (median cycles for code, minus 2 chain cycles): 2.0033

retire uop (01)cycle (02)schedule uop (52)schedule int uop (53)schedule simd uop (54)schedule ldst uop (55)dispatch int uop (56)dispatch simd uop (57)int uops in schedulers (59)ldst uops in schedulers (5b)dispatch uop (78)map int uop (7c)map simd uop (7e)map int uop inputs (7f)map simd uop inputs (81)? int output thing (e9)? simd retires (ee)? int retires (ef)
202044003320101101200000100200003001019248201002002000620060018110000100
202044003320101101200000100200003001019248201002002000420060012110000100
202044070320268112201560111204683091026680206072042058420060012110000100
202044003320101101200000100200003001019248201002002000420060012110000100
202044003320101101200000100200003001019248201002002000420060012110000100
202044003320101101200000100200003001019248201002002000420060012110000100
202044003320101101200000100200003001019248201002002000420060012110000100
202044003320101101200000100200003001019248201002002000420060012110000100
202044003320101101200000100200003001019248201002002000420060012110000100
202044003320101101200000100200003001019248201002002000420060012110000100

1000 unrolls and 10 iterations

Result (median cycles for code, minus 2 chain cycles): 2.0033

retire uop (01)cycle (02)schedule uop (52)schedule int uop (53)schedule simd uop (54)dispatch int uop (56)dispatch simd uop (57)int uops in schedulers (59)ldst uops in schedulers (5b)dispatch uop (78)map int uop (7c)map simd uop (7e)map int uop inputs (7f)map simd uop inputs (81)? int output thing (e9)? simd retires (ee)? int retires (ef)
20024400332001111200001020000301019248200102020006206013211000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010
20024400332001111200001020000301019248200102020000206000011000010

Test 4: Latency 3->3

Code:

  fccmpe s0, s1, #0, lt
  mov x0, 1
  mov x1, 2
  mov x2, 3
  mov x3, 4

(non-fused SUB/CBNZ loop)

100 unrolls and 100 iterations

Result (median cycles for code): 2.0033

retire uop (01)cycle (02)schedule uop (52)schedule int uop (53)schedule simd uop (54)dispatch int uop (56)dispatch simd uop (57)int uops in schedulers (59)ldst uops in schedulers (5b)dispatch uop (78)map int uop (7c)map simd uop (7e)map int uop inputs (7f)map simd uop inputs (81)? int output thing (e9)? int retires (ef)
1020420033102012011000020010000700249768102002001000620030018101100
1020420033102012011000020010000700249769102002001000420030012101100
1020520066102092011000820010021700249769102002001000420030012101100
1020420033102012011000020010000700249769102002001000420030012101100
1020420033102012011000020010000700249769102002001000420030012101100
1020420033102012011000020010000700249769102002001000420030012101100
1020420033102012011000020010000700249769102002001000420030012101100
1020420033102012011000020010000700249769102002001000420030012101100
1020420033102012011000020010000700249769102002001000420030012101100
1020420033102012011000020010000700249769102002001000420030012101100

1000 unrolls and 10 iterations

Result (median cycles for code): 2.0033

retire uop (01)cycle (02)schedule uop (52)schedule int uop (53)schedule simd uop (54)dispatch int uop (56)dispatch simd uop (57)int uops in schedulers (59)ldst uops in schedulers (5b)dispatch uop (78)map int uop (7c)map simd uop (7e)map int uop inputs (7f)map simd uop inputs (81)? int output thing (e9)? int retires (ef)
100242003310021211000020100007024976810020201000620300001110
100242003310021211000020100007024976910020201000020300001110
100242003310021211000020100007024976910020201000020300001110
100242003310021211000020100007024976910020201000020300001110
100242003310021211000020100007024976910020201000020300001110
100242003310021211000020100007024976710020201000420300001110
100242003310021211000020100007024976910020201000020300001110
100242003310021211000020100007024976910020201000020300001110
100242003310021211000020100007024976910020201000020300001110
100242003310021211000020100007024976910020201000020300001110

Test 5: throughput

Count: 8

Code:

  ands xzr, xzr, xzr
  fccmpe s0, s1, #0, lt
  ands xzr, xzr, xzr
  fccmpe s0, s1, #0, lt
  ands xzr, xzr, xzr
  fccmpe s0, s1, #0, lt
  ands xzr, xzr, xzr
  fccmpe s0, s1, #0, lt
  ands xzr, xzr, xzr
  fccmpe s0, s1, #0, lt
  ands xzr, xzr, xzr
  fccmpe s0, s1, #0, lt
  ands xzr, xzr, xzr
  fccmpe s0, s1, #0, lt
  ands xzr, xzr, xzr
  fccmpe s0, s1, #0, lt
  movi v0.16b, 1
  movi v1.16b, 2

(fused SUBS/B.cc loop)

100 unrolls and 100 iterations

Result (median cycles for code divided by count): 1.0736

retire uop (01)cycle (02)schedule uop (52)schedule int uop (53)schedule simd uop (54)dispatch int uop (56)dispatch simd uop (57)int uops in schedulers (59)ldst uops in schedulers (5b)dispatch uop (78)map int uop (7c)map simd uop (7e)map int uop inputs (7f)map simd uop inputs (81)? int output thing (e9)? int retires (ef)
1602048496716015180128800238012780026240321558365160114802078000720024001880008100
1602048395416011180108800038010780006240321560045160114802078000720024001880008100
1602048588616010780105800028010680005240318560031160111802068000620024001880005100
1602048412016014980127800228012680025240318560031160111802068000620024001880005100
1602048588616010780105800028010680005240378476629160151802268002520024001880008100
1602048588616010780105800028010680005240318560031160111802068000620024001880005100
1602048588616010780105800028010680005240378554232160150802268002420024001880005100
1602048588616010780105800028010680005240318560031160111802068000620024001880005100
1602048588616010780105800028010680005240318560031160111802068000620024001880005100
1602048588616010780105800028010680005240318560031160111802068000620024001880005100

1000 unrolls and 10 iterations

Result (median cycles for code divided by count): 1.0483

retire uop (01)cycle (02)schedule uop (52)schedule int uop (53)schedule simd uop (54)dispatch int uop (56)dispatch simd uop (57)int uops in schedulers (59)ldst uops in schedulers (5b)dispatch uop (78)map int uop (7c)map simd uop (7e)map int uop inputs (7f)map simd uop inputs (81)? int output thing (e9)? int retires (ef)
16002485650160018800168000280015800052400455518831600208002580005202400008000110
16002485575160011800118000080010800002400304764251600108002080000202400008000110
16002483862160011800118000080010800002400304764271600108002080000202400008000110
16002483862160011800118000080010800002400304764271600108002080000202400008000110
16002483862160011800118000080010800002400304764291600108002080000202400008000110
16002483865160011800118000080010800002400304764291600108002080000202400008000110
16002483865160011800118000080010800002400304764291600108002080000202400008000110
16002483863160011800118000080010800002400304764291600108002080000202400008000110
16002483863160011800118000080010800002400304764281600108002080000202400008000110
16002483863160011800118000080010800002400304764291600108002080000202400008000110