mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 10:00:16 +02:00
Instruction count CI has transformed the way we work on FEX… I love the system and want to make it better. there’s one part of instruction count CI that isn’t so lovable: the problematic “optimal” flag on instructions. There are several issues with this flag, both philosophical and practical. – it is tedious to update the optimal flag when making an implementation optimal. The effect of that is discouraging people from making instructions, optimal, or encouraging people to fail to update the flag, and dilute the value of it. Either way, since we care far more about optimal implementations, then we do about updating the flag, clearly we should prioritize the implementation and not the flag. This issue was not obvious at the outset, when instruction count, CI was introduced, and still quite small. The problem magnified when we started duplicating instructions in bulk for different combinations of CPU features (flagm, AFP, etc.) that intern multiplies the manual work required to update the flags by the corresponding constant factor. if it comes down to a choice between removing this extra coverage and removing the flag, I think we all agree that removing the flag is the lesser evil. – The definition of “optimal” is fundamentally problematic. I have often improved the instruction count of an instruction that was already “optimal”. This is all kinds of silly, and calls into question whether there’s any value whatsoever in the existing classifications of the flag. Furthermore, it is often unknowable, whether an implementation really is optimal. Is it possible to implement BZHI (with flag calculations) in fewer than eight instructions? We don’t know, and it’s silly to pretend that we do. – as a consequence of the problematic definitions , there are so many errors in both directions that I don’t think there’s much value in preserving the existing classification at the expense of +progress. Being able to say “32% of instructions are translated optimally” is neat, but it really doesn’t tell us anything whatsoever when you dig a little deeper. So, as the flag is misleading at best and perhaps harmful at worst, let’s remove it and make the instruction count CI, more useful overall. let’s let the expected count and the assembly speak for themselves, and cut away the chaff. if we want a meaningless number to report to management, we can instead calculate the average blowup factor ;-) Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
220 lines
4.5 KiB
JSON
220 lines
4.5 KiB
JSON
{
|
|
"Features": {
|
|
"Bitness": 64,
|
|
"EnabledHostFeatures": [
|
|
"AFP"
|
|
],
|
|
"DisabledHostFeatures": [
|
|
"SVE128",
|
|
"SVE256"
|
|
]
|
|
},
|
|
"Instructions": {
|
|
"cvtsi2sd xmm0, eax": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x2a"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"scvtf d16, w4"
|
|
]
|
|
},
|
|
"cvtsi2sd xmm0, dword [rax]": {
|
|
"ExpectedInstructionCount": 2,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x2a"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"ldr w20, [x4]",
|
|
"scvtf d16, w20"
|
|
]
|
|
},
|
|
"cvtsi2sd xmm0, rax": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x2a"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"scvtf d16, x4"
|
|
]
|
|
},
|
|
"cvtsi2sd xmm0, qword [rax]": {
|
|
"ExpectedInstructionCount": 2,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x2a"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"ldr d2, [x4]",
|
|
"scvtf d16, d2"
|
|
]
|
|
},
|
|
"sqrtsd xmm0, xmm1": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x51"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fsqrt d16, d17"
|
|
]
|
|
},
|
|
"addsd xmm0, xmm1": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x58"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fadd d16, d16, d17"
|
|
]
|
|
},
|
|
"mulsd xmm0, xmm1": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x59"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fmul d16, d16, d17"
|
|
]
|
|
},
|
|
"cvtsd2ss xmm0, xmm1": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x5a"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fcvt s16, d17"
|
|
]
|
|
},
|
|
"cvtsd2ss xmm0, [rax]": {
|
|
"ExpectedInstructionCount": 2,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x5a"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"ldr q2, [x4]",
|
|
"fcvt s16, d2"
|
|
]
|
|
},
|
|
"subsd xmm0, xmm1": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x5c"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fsub d16, d16, d17"
|
|
]
|
|
},
|
|
"minsd xmm0, xmm1": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x5d"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fmin d16, d16, d17"
|
|
]
|
|
},
|
|
"divsd xmm0, xmm1": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x5e"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fdiv d16, d16, d17"
|
|
]
|
|
},
|
|
"maxsd xmm0, xmm1": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0x5f"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fmax d16, d16, d17"
|
|
]
|
|
},
|
|
"cmpsd xmm0, xmm1, 0": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0xc2"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fcmeq d16, d16, d17"
|
|
]
|
|
},
|
|
"cmpsd xmm0, xmm1, 1": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0xc2"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fcmgt d16, d17, d16"
|
|
]
|
|
},
|
|
"cmpsd xmm0, xmm1, 2": {
|
|
"ExpectedInstructionCount": 1,
|
|
"Comment": [
|
|
"0xf2 0x0f 0xc2"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fcmge d16, d17, d16"
|
|
]
|
|
},
|
|
"cmpsd xmm0, xmm1, 3": {
|
|
"ExpectedInstructionCount": 5,
|
|
"Comment": [
|
|
"0xf2 0x0f 0xc2"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fcmge d0, d16, d17",
|
|
"fcmgt d1, d17, d16",
|
|
"orr v0.8b, v0.8b, v1.8b",
|
|
"mvn v0.8b, v0.8b",
|
|
"mov v16.d[0], v0.d[0]"
|
|
]
|
|
},
|
|
"cmpsd xmm0, xmm1, 4": {
|
|
"ExpectedInstructionCount": 3,
|
|
"Comment": [
|
|
"0xf2 0x0f 0xc2"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fcmeq d0, d16, d17",
|
|
"mvn v0.8b, v0.8b",
|
|
"mov v16.d[0], v0.d[0]"
|
|
]
|
|
},
|
|
"cmpsd xmm0, xmm1, 5": {
|
|
"ExpectedInstructionCount": 3,
|
|
"Comment": [
|
|
"0xf2 0x0f 0xc2"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fcmgt d2, d17, d16",
|
|
"mvn v2.16b, v2.16b",
|
|
"mov v16.d[0], v2.d[0]"
|
|
]
|
|
},
|
|
"cmpsd xmm0, xmm1, 6": {
|
|
"ExpectedInstructionCount": 3,
|
|
"Comment": [
|
|
"0xf2 0x0f 0xc2"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fcmge d2, d17, d16",
|
|
"mvn v2.16b, v2.16b",
|
|
"mov v16.d[0], v2.d[0]"
|
|
]
|
|
},
|
|
"cmpsd xmm0, xmm1, 7": {
|
|
"ExpectedInstructionCount": 4,
|
|
"Comment": [
|
|
"0xf2 0x0f 0xc2"
|
|
],
|
|
"ExpectedArm64ASM": [
|
|
"fcmge d0, d16, d17",
|
|
"fcmgt d1, d17, d16",
|
|
"orr v0.8b, v0.8b, v1.8b",
|
|
"mov v16.d[0], v0.d[0]"
|
|
]
|
|
}
|
|
}
|
|
}
|