mirror of
https://github.com/wazero/wazero
synced 2026-06-21 14:12:37 +00:00
wazevo: passes simd_boolean spec tests (#1724)
Signed-off-by: Edoardo Vacchi <evacchi@users.noreply.github.com>
This commit is contained in:
@@ -99,8 +99,12 @@ var defKinds = [numInstructionKinds]defKind{
|
||||
fpuCSel: defKindRD,
|
||||
movToVec: defKindRD,
|
||||
movFromVec: defKindRD,
|
||||
vecDup: defKindRD,
|
||||
vecExtract: defKindRD,
|
||||
vecMisc: defKindRD,
|
||||
vecLanes: defKindRD,
|
||||
vecShiftImm: defKindRD,
|
||||
vecPermute: defKindRD,
|
||||
vecRRR: defKindRD,
|
||||
fpuToInt: defKindRD,
|
||||
intToFpu: defKindRD,
|
||||
@@ -205,10 +209,14 @@ var useKinds = [numInstructionKinds]useKind{
|
||||
fpuCSel: useKindRNRM,
|
||||
movToVec: useKindRN,
|
||||
movFromVec: useKindRN,
|
||||
vecDup: useKindRN,
|
||||
vecExtract: useKindRNRM,
|
||||
cCmpImm: useKindRN,
|
||||
vecMisc: useKindRN,
|
||||
vecLanes: useKindRN,
|
||||
vecShiftImm: useKindRN,
|
||||
vecRRR: useKindRNRM,
|
||||
vecPermute: useKindRNRM,
|
||||
fpuToInt: useKindRN,
|
||||
intToFpu: useKindRN,
|
||||
movToFPSR: useKindRN,
|
||||
@@ -775,6 +783,19 @@ func (i *instruction) asMovFromVec(rd, rn operand, arr vecArrangement, index vec
|
||||
i.u1, i.u2 = uint64(arr), uint64(index)
|
||||
}
|
||||
|
||||
func (i *instruction) asVecDup(rd, rn operand, arr vecArrangement) {
|
||||
i.kind = vecDup
|
||||
i.u1 = uint64(arr)
|
||||
i.rn, i.rd = rn, rd
|
||||
}
|
||||
|
||||
func (i *instruction) asVecExtract(rd, rn, rm operand, arr vecArrangement, index uint32) {
|
||||
i.kind = vecExtract
|
||||
i.u1 = uint64(arr)
|
||||
i.rn, i.rm, i.rd = rn, rm, rd
|
||||
i.u2 = uint64(index)
|
||||
}
|
||||
|
||||
func (i *instruction) asVecMisc(op vecOp, rd, rn operand, arr vecArrangement) {
|
||||
i.kind = vecMisc
|
||||
i.u1 = uint64(op)
|
||||
@@ -789,6 +810,20 @@ func (i *instruction) asVecLanes(op vecOp, rd, rn operand, arr vecArrangement) {
|
||||
i.u2 = uint64(arr)
|
||||
}
|
||||
|
||||
func (i *instruction) asVecShiftImm(op vecOp, rd, rn, rm operand, arr vecArrangement) {
|
||||
i.kind = vecShiftImm
|
||||
i.u1 = uint64(op)
|
||||
i.rn, i.rm, i.rd = rn, rm, rd
|
||||
i.u2 = uint64(arr)
|
||||
}
|
||||
|
||||
func (i *instruction) asVecPermute(op vecOp, rd, rn, rm operand, arr vecArrangement) {
|
||||
i.kind = vecPermute
|
||||
i.u1 = uint64(op)
|
||||
i.rn, i.rm, i.rd = rn, rm, rd
|
||||
i.u2 = uint64(arr)
|
||||
}
|
||||
|
||||
func (i *instruction) asVecRRR(op vecOp, rd, rn, rm operand, arr vecArrangement) {
|
||||
i.kind = vecRRR
|
||||
i.u1 = uint64(op)
|
||||
@@ -1076,9 +1111,19 @@ func (i *instruction) String() (str string) {
|
||||
case movFromVecSigned:
|
||||
panic("TODO")
|
||||
case vecDup:
|
||||
panic("TODO")
|
||||
str = fmt.Sprintf("dup %s, %s",
|
||||
formatVRegVec(i.rd.nr(), vecArrangement(i.u1), vecIndexNone),
|
||||
formatVRegSized(i.rn.nr(), 64),
|
||||
)
|
||||
case vecDupFromFpu:
|
||||
panic("TODO")
|
||||
case vecExtract:
|
||||
str = fmt.Sprintf("ext %s, %s, %s, #%d",
|
||||
formatVRegVec(i.rd.nr(), vecArrangement(i.u1), vecIndexNone),
|
||||
formatVRegVec(i.rn.nr(), vecArrangement(i.u1), vecIndexNone),
|
||||
formatVRegVec(i.rm.nr(), vecArrangement(i.u1), vecIndexNone),
|
||||
uint32(i.u2),
|
||||
)
|
||||
case vecExtend:
|
||||
panic("TODO")
|
||||
case vecMovElement:
|
||||
@@ -1093,10 +1138,17 @@ func (i *instruction) String() (str string) {
|
||||
formatVRegVec(i.rm.nr(), vecArrangement(i.u2), vecIndexNone),
|
||||
)
|
||||
case vecMisc:
|
||||
str = fmt.Sprintf("%s %s, %s",
|
||||
vecOp(i.u1),
|
||||
formatVRegVec(i.rd.nr(), vecArrangement(i.u2), vecIndexNone),
|
||||
formatVRegVec(i.rn.nr(), vecArrangement(i.u2), vecIndexNone))
|
||||
vop := vecOp(i.u1)
|
||||
if vop == vecOpCmeq0 {
|
||||
str = fmt.Sprintf("cmeq %s, %s, #0",
|
||||
formatVRegVec(i.rd.nr(), vecArrangement(i.u2), vecIndexNone),
|
||||
formatVRegVec(i.rn.nr(), vecArrangement(i.u2), vecIndexNone))
|
||||
} else {
|
||||
str = fmt.Sprintf("%s %s, %s",
|
||||
vop,
|
||||
formatVRegVec(i.rd.nr(), vecArrangement(i.u2), vecIndexNone),
|
||||
formatVRegVec(i.rn.nr(), vecArrangement(i.u2), vecIndexNone))
|
||||
}
|
||||
case vecLanes:
|
||||
arr := vecArrangement(i.u2)
|
||||
var destArr vecArrangement
|
||||
@@ -1114,10 +1166,24 @@ func (i *instruction) String() (str string) {
|
||||
vecOp(i.u1),
|
||||
formatVRegWidthVec(i.rd.nr(), destArr),
|
||||
formatVRegVec(i.rn.nr(), arr, vecIndexNone))
|
||||
case vecShiftImm:
|
||||
arr := vecArrangement(i.u2)
|
||||
str = fmt.Sprintf("%s %s, %s, #%d",
|
||||
vecOp(i.u1),
|
||||
formatVRegVec(i.rd.nr(), arr, vecIndexNone),
|
||||
formatVRegVec(i.rn.nr(), arr, vecIndexNone),
|
||||
i.rm.shiftImm())
|
||||
case vecTbl:
|
||||
panic("TODO")
|
||||
case vecTbl2:
|
||||
panic("TODO")
|
||||
case vecPermute:
|
||||
arr := vecArrangement(i.u2)
|
||||
str = fmt.Sprintf("%s %s, %s, %s",
|
||||
vecOp(i.u1),
|
||||
formatVRegVec(i.rd.nr(), arr, vecIndexNone),
|
||||
formatVRegVec(i.rn.nr(), arr, vecIndexNone),
|
||||
formatVRegVec(i.rm.nr(), arr, vecIndexNone))
|
||||
case movToFPSR:
|
||||
str = fmt.Sprintf("msr fpsr, %s", formatVRegSized(i.rn.nr(), 64))
|
||||
case movFromFPSR:
|
||||
@@ -1322,6 +1388,8 @@ const (
|
||||
vecDup
|
||||
// vecDupFromFpu represents a duplication of scalar to vector.
|
||||
vecDupFromFpu
|
||||
// vecExtract represents a vector extraction operation.
|
||||
vecExtract
|
||||
// vecExtend represents a vector extension operation.
|
||||
vecExtend
|
||||
// vecMovElement represents a move vector element to another vector element operation.
|
||||
@@ -1334,10 +1402,14 @@ const (
|
||||
vecMisc
|
||||
// vecLanes represents a vector instruction across lanes.
|
||||
vecLanes
|
||||
// vecShiftImm represents a SIMD scalar shift by immediate instruction.
|
||||
vecShiftImm
|
||||
// vecTbl represents a table vector lookup - single register table.
|
||||
vecTbl
|
||||
// vecTbl2 represents a table vector lookup - two register table.
|
||||
vecTbl2
|
||||
// vecPermute represents a vector permute instruction.
|
||||
vecPermute
|
||||
// movToNZCV represents a move to the FPSR.
|
||||
movToFPSR
|
||||
// movFromNZCV represents a move from the FPSR.
|
||||
@@ -1502,6 +1574,8 @@ func (b vecOp) String() string {
|
||||
switch b {
|
||||
case vecOpCnt:
|
||||
return "cnt"
|
||||
case vecOpCmeq0:
|
||||
return "cmeq0"
|
||||
case vecOpUaddlv:
|
||||
return "uaddlv"
|
||||
case vecOpBit:
|
||||
@@ -1522,16 +1596,22 @@ func (b vecOp) String() string {
|
||||
return "add"
|
||||
case vecOpAddp:
|
||||
return "addp"
|
||||
case vecOpAddv:
|
||||
return "addv"
|
||||
case vecOpSub:
|
||||
return "sub"
|
||||
case vecOpSmin:
|
||||
return "smin"
|
||||
case vecOpUmin:
|
||||
return "umin"
|
||||
case vecOpUminv:
|
||||
return "uminv"
|
||||
case vecOpSmax:
|
||||
return "smax"
|
||||
case vecOpUmax:
|
||||
return "umax"
|
||||
case vecOpUmaxp:
|
||||
return "umaxp"
|
||||
case vecOpUrhadd:
|
||||
return "urhadd"
|
||||
case vecOpMul:
|
||||
@@ -1546,12 +1626,17 @@ func (b vecOp) String() string {
|
||||
return "xtn"
|
||||
case vecOpShll:
|
||||
return "shll"
|
||||
case vecOpSshr:
|
||||
return "sshr"
|
||||
case vecOpZip1:
|
||||
return "zip1"
|
||||
}
|
||||
panic(int(b))
|
||||
}
|
||||
|
||||
const (
|
||||
vecOpCnt vecOp = iota
|
||||
vecOpCmeq0
|
||||
vecOpUaddlv
|
||||
vecOpBit
|
||||
vecOpBic
|
||||
@@ -1561,6 +1646,7 @@ const (
|
||||
vecOpOrr
|
||||
vecOpEOR
|
||||
vecOpAdd
|
||||
vecOpAddv
|
||||
vecOpSqadd
|
||||
vecOpUqadd
|
||||
vecOpAddp
|
||||
@@ -1569,6 +1655,7 @@ const (
|
||||
vecOpUqsub
|
||||
vecOpSmin
|
||||
vecOpUmin
|
||||
vecOpUminv
|
||||
vecOpSmax
|
||||
vecOpUmax
|
||||
vecOpUmaxp
|
||||
@@ -1580,6 +1667,8 @@ const (
|
||||
vecOpRev64
|
||||
vecOpXtn
|
||||
vecOpShll
|
||||
vecOpSshr
|
||||
vecOpZip1
|
||||
)
|
||||
|
||||
// bitOp determines the type of bitwise operation. Instructions whose kind is one of
|
||||
|
||||
@@ -262,6 +262,25 @@ func (i *instruction) encode(c backend.Compiler) {
|
||||
vecArrangement(byte(i.u1)),
|
||||
vecIndex(i.u2),
|
||||
))
|
||||
case vecDup:
|
||||
c.Emit4Bytes(encodeVecDup(
|
||||
regNumberInEncoding[i.rd.realReg()],
|
||||
regNumberInEncoding[i.rn.realReg()],
|
||||
vecArrangement(byte(i.u1))))
|
||||
case vecExtract:
|
||||
c.Emit4Bytes(encodeVecExtract(
|
||||
regNumberInEncoding[i.rd.realReg()],
|
||||
regNumberInEncoding[i.rn.realReg()],
|
||||
regNumberInEncoding[i.rm.realReg()],
|
||||
vecArrangement(byte(i.u1)),
|
||||
uint32(i.u2)))
|
||||
case vecPermute:
|
||||
c.Emit4Bytes(encodeVecPermute(
|
||||
vecOp(i.u1),
|
||||
regNumberInEncoding[i.rd.realReg()],
|
||||
regNumberInEncoding[i.rn.realReg()],
|
||||
regNumberInEncoding[i.rm.realReg()],
|
||||
vecArrangement(byte(i.u2))))
|
||||
case vecMisc:
|
||||
c.Emit4Bytes(encodeAdvancedSIMDTwoMisc(
|
||||
vecOp(i.u1),
|
||||
@@ -277,6 +296,14 @@ func (i *instruction) encode(c backend.Compiler) {
|
||||
regNumberInEncoding[i.rn.realReg()],
|
||||
vecArrangement(i.u2),
|
||||
))
|
||||
case vecShiftImm:
|
||||
c.Emit4Bytes(encodeVecShiftImm(
|
||||
vecOp(i.u1),
|
||||
regNumberInEncoding[i.rd.realReg()],
|
||||
regNumberInEncoding[i.rn.realReg()],
|
||||
uint32(i.rm.shiftImm()),
|
||||
vecArrangement(i.u2),
|
||||
))
|
||||
case brTableSequence:
|
||||
encodeBrTableSequence(c, i.rn.reg(), i.targets)
|
||||
case fpuToInt, intToFpu:
|
||||
@@ -717,6 +744,66 @@ func encodeMoveFromVec(rd, rn uint32, arr vecArrangement, index vecIndex) uint32
|
||||
return 0b0_001110000<<21 | q<<30 | imm5<<16 | 0b001111<<10 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// encodeVecDup encodes as "Duplicate general-purpose register to vector."
|
||||
// (represented as `dup`)
|
||||
// https://developer.arm.com/documentation/ddi0596/2020-12/SIMD-FP-Instructions/DUP--general---Duplicate-general-purpose-register-to-vector-?lang=en
|
||||
func encodeVecDup(rd, rn uint32, arr vecArrangement) uint32 {
|
||||
var q, imm5 uint32
|
||||
switch arr {
|
||||
case vecArrangement8B:
|
||||
q, imm5 = 0b0, 0b1
|
||||
case vecArrangement16B:
|
||||
q, imm5 = 0b1, 0b1
|
||||
case vecArrangement4H:
|
||||
q, imm5 = 0b0, 0b10
|
||||
case vecArrangement8H:
|
||||
q, imm5 = 0b1, 0b10
|
||||
case vecArrangement2S:
|
||||
q, imm5 = 0b0, 0b100
|
||||
case vecArrangement4S:
|
||||
q, imm5 = 0b1, 0b100
|
||||
case vecArrangement2D:
|
||||
q, imm5 = 0b1, 0b1000
|
||||
default:
|
||||
panic("Unsupported arrangement " + arr.String())
|
||||
}
|
||||
return q<<30 | 0b001110000<<21 | imm5<<16 | 0b000011<<10 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// encodeVecExtract encodes as "Advanced SIMD extract."
|
||||
// Currently only `ext` is defined.
|
||||
// https://developer.arm.com/documentation/ddi0602/2023-06/Index-by-Encoding/Data-Processing----Scalar-Floating-Point-and-Advanced-SIMD?lang=en#simd-dp
|
||||
// https://developer.arm.com/documentation/ddi0602/2023-06/SIMD-FP-Instructions/EXT--Extract-vector-from-pair-of-vectors-?lang=en
|
||||
func encodeVecExtract(rd, rn, rm uint32, arr vecArrangement, index uint32) uint32 {
|
||||
var q, imm4 uint32
|
||||
switch arr {
|
||||
case vecArrangement8B:
|
||||
q, imm4 = 0, 0b0111&uint32(index)
|
||||
case vecArrangement16B:
|
||||
q, imm4 = 1, 0b1111&uint32(index)
|
||||
default:
|
||||
panic("Unsupported arrangement " + arr.String())
|
||||
}
|
||||
return q<<30 | 0b101110000<<21 | rm<<16 | imm4<<11 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// encodeVecPermute encodes as "Advanced SIMD permute."
|
||||
// https://developer.arm.com/documentation/ddi0602/2023-06/Index-by-Encoding/Data-Processing----Scalar-Floating-Point-and-Advanced-SIMD?lang=en#simd-dp
|
||||
func encodeVecPermute(op vecOp, rd, rn, rm uint32, arr vecArrangement) uint32 {
|
||||
var q, size, opcode uint32
|
||||
switch op {
|
||||
case vecOpZip1:
|
||||
opcode = 0b011
|
||||
if arr == vecArrangement1D {
|
||||
panic("unsupported arrangement: " + arr.String())
|
||||
}
|
||||
size, q = arrToSizeQEncoded(arr)
|
||||
default:
|
||||
panic("TODO: " + op.String())
|
||||
}
|
||||
return q<<30 | 0b001110<<24 | size<<22 | rm<<16 | opcode<<12 | 0b10<<10 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// encodeConditionalSelect encodes as "Conditional select" in
|
||||
// https://developer.arm.com/documentation/ddi0596/2020-12/Index-by-Encoding/Data-Processing----Register?lang=en#condsel
|
||||
func encodeConditionalSelect(kind instructionKind, rd, rn, rm uint32, c condFlag, _64bit bool) uint32 {
|
||||
@@ -1434,29 +1521,76 @@ func encodeAluRRImm(op aluOp, rd, rn, amount, _64bit uint32) uint32 {
|
||||
// https://developer.arm.com/documentation/ddi0596/2020-12/Index-by-Encoding/Data-Processing----Scalar-Floating-Point-and-Advanced-SIMD?lang=en
|
||||
func encodeVecLanes(op vecOp, rd uint32, rn uint32, arr vecArrangement) uint32 {
|
||||
var u, q, size, opcode uint32
|
||||
switch arr {
|
||||
case vecArrangement8B:
|
||||
q, size = 0b0, 0b00
|
||||
case vecArrangement16B:
|
||||
q, size = 0b1, 0b00
|
||||
case vecArrangement4H:
|
||||
q, size = 0, 0b01
|
||||
case vecArrangement8H:
|
||||
q, size = 1, 0b01
|
||||
case vecArrangement4S:
|
||||
q, size = 1, 0b10
|
||||
default:
|
||||
panic("unsupported arrangement: " + arr.String())
|
||||
}
|
||||
switch op {
|
||||
case vecOpUaddlv:
|
||||
u, opcode = 1, 0b00011
|
||||
switch arr {
|
||||
case vecArrangement8B:
|
||||
q, size = 0b0, 0b00
|
||||
case vecArrangement16B:
|
||||
q, size = 0b1, 0b00
|
||||
case vecArrangement4H:
|
||||
q, size = 0, 0b01
|
||||
case vecArrangement8H:
|
||||
q, size = 1, 0b01
|
||||
case vecArrangement4S:
|
||||
q, size = 1, 0b10
|
||||
default:
|
||||
panic("unsupported arrangement: " + arr.String())
|
||||
}
|
||||
case vecOpUminv:
|
||||
u, opcode = 1, 0b11010
|
||||
case vecOpAddv:
|
||||
u, opcode = 0, 0b11011
|
||||
default:
|
||||
panic("unsupported or illegal vecOp: " + op.String())
|
||||
}
|
||||
return q<<30 | u<<29 | 0b1110<<24 | size<<22 | 0b11000<<17 | opcode<<12 | 0b10<<10 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// encodeVecLanes encodes as Data Processing (Advanced SIMD scalar shift by immediate) depending on vecOp in
|
||||
// https://developer.arm.com/documentation/ddi0596/2020-12/Index-by-Encoding/Data-Processing----Scalar-Floating-Point-and-Advanced-SIMD?lang=en
|
||||
func encodeVecShiftImm(op vecOp, rd uint32, rn, amount uint32, arr vecArrangement) uint32 {
|
||||
var u, q, immh, immb, opcode uint32
|
||||
switch op {
|
||||
case vecOpSshr:
|
||||
u, opcode = 0, 0b00000
|
||||
switch arr {
|
||||
case vecArrangement16B:
|
||||
q = 0b1
|
||||
fallthrough
|
||||
case vecArrangement8B:
|
||||
immh = 0b0001
|
||||
immb = 8 - uint32(amount&0b111)
|
||||
case vecArrangement8H:
|
||||
q = 0b1
|
||||
fallthrough
|
||||
case vecArrangement4H:
|
||||
v := 16 - uint32(amount&0b1111)
|
||||
immb = v & 0b111
|
||||
immh = 0b0010 | (v >> 3)
|
||||
case vecArrangement4S:
|
||||
q = 0b1
|
||||
fallthrough
|
||||
case vecArrangement2S:
|
||||
v := 32 - uint32(amount&0b11111)
|
||||
immb = v & 0b111
|
||||
immh = 0b0100 | (v >> 3)
|
||||
case vecArrangement2D:
|
||||
q = 0b1
|
||||
v := 64 - uint32(amount&0b111111)
|
||||
immb = v & 0b111
|
||||
immh = 0b1000 | (v >> 3)
|
||||
default:
|
||||
panic("unsupported arrangement: " + arr.String())
|
||||
}
|
||||
|
||||
default:
|
||||
panic("unsupported or illegal vecOp: " + op.String())
|
||||
}
|
||||
return q<<30 | u<<29 | 0b011110<<23 | immh<<19 | immb<<16 | 0b000001<<10 | opcode<<11 | 0b1<<10 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// encodeVecMisc encodes as Data Processing (Advanced SIMD two-register miscellaneous) depending on vecOp in
|
||||
// https://developer.arm.com/documentation/ddi0596/2020-12/Index-by-Encoding/Data-Processing----Scalar-Floating-Point-and-Advanced-SIMD?lang=en#simd-dp
|
||||
func encodeAdvancedSIMDTwoMisc(op vecOp, rd, rn uint32, arr vecArrangement) uint32 {
|
||||
@@ -1472,6 +1606,12 @@ func encodeAdvancedSIMDTwoMisc(op vecOp, rd, rn uint32, arr vecArrangement) uint
|
||||
default:
|
||||
panic("unsupported arrangement: " + arr.String())
|
||||
}
|
||||
case vecOpCmeq0:
|
||||
if arr == vecArrangement1D {
|
||||
panic("unsupported arrangement: " + arr.String())
|
||||
}
|
||||
opcode = 0b01001
|
||||
size, q = arrToSizeQEncoded(arr)
|
||||
case vecOpNot:
|
||||
u = 1
|
||||
opcode = 0b00101
|
||||
|
||||
@@ -27,6 +27,61 @@ func TestInstruction_encode(t *testing.T) {
|
||||
{want: "21443bd5", setup: func(i *instruction) { i.asMovFromFPSR(x1VReg) }},
|
||||
{want: "2f08417a", setup: func(i *instruction) { i.asCCmpImm(operandNR(x1VReg), 1, eq, 0b1111, false) }},
|
||||
{want: "201841fa", setup: func(i *instruction) { i.asCCmpImm(operandNR(x1VReg), 1, ne, 0, true) }},
|
||||
{want: "410c010e", setup: func(i *instruction) { i.asVecDup(operandNR(v1VReg), operandNR(v2VReg), vecArrangement8B) }},
|
||||
{want: "410c014e", setup: func(i *instruction) { i.asVecDup(operandNR(v1VReg), operandNR(v2VReg), vecArrangement16B) }},
|
||||
{want: "410c020e", setup: func(i *instruction) { i.asVecDup(operandNR(v1VReg), operandNR(v2VReg), vecArrangement4H) }},
|
||||
{want: "410c024e", setup: func(i *instruction) { i.asVecDup(operandNR(v1VReg), operandNR(v2VReg), vecArrangement8H) }},
|
||||
{want: "410c040e", setup: func(i *instruction) { i.asVecDup(operandNR(v1VReg), operandNR(v2VReg), vecArrangement2S) }},
|
||||
{want: "410c044e", setup: func(i *instruction) { i.asVecDup(operandNR(v1VReg), operandNR(v2VReg), vecArrangement4S) }},
|
||||
{want: "410c084e", setup: func(i *instruction) { i.asVecDup(operandNR(v1VReg), operandNR(v2VReg), vecArrangement2D) }},
|
||||
{want: "4138032e", setup: func(i *instruction) {
|
||||
i.asVecExtract(operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement8B, 7)
|
||||
}},
|
||||
{want: "4138036e", setup: func(i *instruction) {
|
||||
i.asVecExtract(operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement16B, 7)
|
||||
}},
|
||||
{want: "4104090f", setup: func(i *instruction) {
|
||||
i.asVecShiftImm(vecOpSshr, operandNR(v1VReg), operandNR(v2VReg), operandShiftImm(7), vecArrangement8B)
|
||||
}},
|
||||
{want: "4104094f", setup: func(i *instruction) {
|
||||
i.asVecShiftImm(vecOpSshr, operandNR(v1VReg), operandNR(v2VReg), operandShiftImm(7), vecArrangement16B)
|
||||
}},
|
||||
{want: "4104190f", setup: func(i *instruction) {
|
||||
i.asVecShiftImm(vecOpSshr, operandNR(v1VReg), operandNR(v2VReg), operandShiftImm(7), vecArrangement4H)
|
||||
}},
|
||||
{want: "4104194f", setup: func(i *instruction) {
|
||||
i.asVecShiftImm(vecOpSshr, operandNR(v1VReg), operandNR(v2VReg), operandShiftImm(7), vecArrangement8H)
|
||||
}},
|
||||
{want: "4104390f", setup: func(i *instruction) {
|
||||
i.asVecShiftImm(vecOpSshr, operandNR(v1VReg), operandNR(v2VReg), operandShiftImm(7), vecArrangement2S)
|
||||
}},
|
||||
{want: "4104394f", setup: func(i *instruction) {
|
||||
i.asVecShiftImm(vecOpSshr, operandNR(v1VReg), operandNR(v2VReg), operandShiftImm(7), vecArrangement4S)
|
||||
}},
|
||||
{want: "4104794f", setup: func(i *instruction) {
|
||||
i.asVecShiftImm(vecOpSshr, operandNR(v1VReg), operandNR(v2VReg), operandShiftImm(7), vecArrangement2D)
|
||||
}},
|
||||
{want: "4138030e", setup: func(i *instruction) {
|
||||
i.asVecPermute(vecOpZip1, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement8B)
|
||||
}},
|
||||
{want: "4138034e", setup: func(i *instruction) {
|
||||
i.asVecPermute(vecOpZip1, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement16B)
|
||||
}},
|
||||
{want: "4138430e", setup: func(i *instruction) {
|
||||
i.asVecPermute(vecOpZip1, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement4H)
|
||||
}},
|
||||
{want: "4138434e", setup: func(i *instruction) {
|
||||
i.asVecPermute(vecOpZip1, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement8H)
|
||||
}},
|
||||
{want: "4138830e", setup: func(i *instruction) {
|
||||
i.asVecPermute(vecOpZip1, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement2S)
|
||||
}},
|
||||
{want: "4138834e", setup: func(i *instruction) {
|
||||
i.asVecPermute(vecOpZip1, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement4S)
|
||||
}},
|
||||
{want: "4138c34e", setup: func(i *instruction) {
|
||||
i.asVecPermute(vecOpZip1, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement2D)
|
||||
}},
|
||||
{want: "411ca32e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpBit, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement8B)
|
||||
}},
|
||||
@@ -174,6 +229,21 @@ func TestInstruction_encode(t *testing.T) {
|
||||
{want: "41bce34e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpAddp, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement2D)
|
||||
}},
|
||||
{want: "41bc230e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpAddp, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement8B)
|
||||
}},
|
||||
{want: "41b8314e", setup: func(i *instruction) {
|
||||
i.asVecLanes(vecOpAddv, operandNR(v1VReg), operandNR(v2VReg), vecArrangement16B)
|
||||
}},
|
||||
{want: "41b8710e", setup: func(i *instruction) {
|
||||
i.asVecLanes(vecOpAddv, operandNR(v1VReg), operandNR(v2VReg), vecArrangement4H)
|
||||
}},
|
||||
{want: "41b8714e", setup: func(i *instruction) {
|
||||
i.asVecLanes(vecOpAddv, operandNR(v1VReg), operandNR(v2VReg), vecArrangement8H)
|
||||
}},
|
||||
{want: "41b8b14e", setup: func(i *instruction) {
|
||||
i.asVecLanes(vecOpAddv, operandNR(v1VReg), operandNR(v2VReg), vecArrangement4S)
|
||||
}},
|
||||
{want: "416c230e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpSmin, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement8B)
|
||||
}},
|
||||
@@ -246,6 +316,36 @@ func TestInstruction_encode(t *testing.T) {
|
||||
{want: "4164a36e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpUmax, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement4S)
|
||||
}},
|
||||
{want: "41a4232e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpUmaxp, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement8B)
|
||||
}},
|
||||
{want: "41a4236e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpUmaxp, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement16B)
|
||||
}},
|
||||
{want: "41a4632e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpUmaxp, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement4H)
|
||||
}},
|
||||
{want: "41a4636e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpUmaxp, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement8H)
|
||||
}},
|
||||
{want: "41a4a32e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpUmaxp, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement2S)
|
||||
}},
|
||||
{want: "41a8312e", setup: func(i *instruction) {
|
||||
i.asVecLanes(vecOpUminv, operandNR(v1VReg), operandNR(v2VReg), vecArrangement8B)
|
||||
}},
|
||||
{want: "41a8316e", setup: func(i *instruction) {
|
||||
i.asVecLanes(vecOpUminv, operandNR(v1VReg), operandNR(v2VReg), vecArrangement16B)
|
||||
}},
|
||||
{want: "41a8712e", setup: func(i *instruction) {
|
||||
i.asVecLanes(vecOpUminv, operandNR(v1VReg), operandNR(v2VReg), vecArrangement4H)
|
||||
}},
|
||||
{want: "41a8716e", setup: func(i *instruction) {
|
||||
i.asVecLanes(vecOpUminv, operandNR(v1VReg), operandNR(v2VReg), vecArrangement8H)
|
||||
}},
|
||||
{want: "41a8b16e", setup: func(i *instruction) {
|
||||
i.asVecLanes(vecOpUminv, operandNR(v1VReg), operandNR(v2VReg), vecArrangement4S)
|
||||
}},
|
||||
{want: "4114232e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpUrhadd, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement8B)
|
||||
}},
|
||||
@@ -282,6 +382,27 @@ func TestInstruction_encode(t *testing.T) {
|
||||
{want: "419ca34e", setup: func(i *instruction) {
|
||||
i.asVecRRR(vecOpMul, operandNR(v1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement4S)
|
||||
}},
|
||||
{want: "4198200e", setup: func(i *instruction) {
|
||||
i.asVecMisc(vecOpCmeq0, operandNR(v1VReg), operandNR(v2VReg), vecArrangement8B)
|
||||
}},
|
||||
{want: "4198204e", setup: func(i *instruction) {
|
||||
i.asVecMisc(vecOpCmeq0, operandNR(v1VReg), operandNR(v2VReg), vecArrangement16B)
|
||||
}},
|
||||
{want: "4198600e", setup: func(i *instruction) {
|
||||
i.asVecMisc(vecOpCmeq0, operandNR(v1VReg), operandNR(v2VReg), vecArrangement4H)
|
||||
}},
|
||||
{want: "4198604e", setup: func(i *instruction) {
|
||||
i.asVecMisc(vecOpCmeq0, operandNR(v1VReg), operandNR(v2VReg), vecArrangement8H)
|
||||
}},
|
||||
{want: "4198a00e", setup: func(i *instruction) {
|
||||
i.asVecMisc(vecOpCmeq0, operandNR(v1VReg), operandNR(v2VReg), vecArrangement2S)
|
||||
}},
|
||||
{want: "4198a04e", setup: func(i *instruction) {
|
||||
i.asVecMisc(vecOpCmeq0, operandNR(v1VReg), operandNR(v2VReg), vecArrangement4S)
|
||||
}},
|
||||
{want: "4198e04e", setup: func(i *instruction) {
|
||||
i.asVecMisc(vecOpCmeq0, operandNR(v1VReg), operandNR(v2VReg), vecArrangement2D)
|
||||
}},
|
||||
{want: "41b8200e", setup: func(i *instruction) {
|
||||
i.asVecMisc(vecOpAbs, operandNR(v1VReg), operandNR(v2VReg), vecArrangement8B)
|
||||
}},
|
||||
@@ -346,6 +467,16 @@ func TestInstruction_encode(t *testing.T) {
|
||||
{want: "413c020e", setup: func(i *instruction) { i.asMovFromVec(operandNR(x1VReg), operandNR(v2VReg), vecArrangementH, 0) }},
|
||||
{want: "413c040e", setup: func(i *instruction) { i.asMovFromVec(operandNR(x1VReg), operandNR(v2VReg), vecArrangementS, 0) }},
|
||||
{want: "413c084e", setup: func(i *instruction) { i.asMovFromVec(operandNR(x1VReg), operandNR(v2VReg), vecArrangementD, 0) }},
|
||||
{want: "410c084e", setup: func(i *instruction) { i.asVecDup(operandNR(x1VReg), operandNR(v2VReg), vecArrangement2D) }},
|
||||
{want: "4140036e", setup: func(i *instruction) { // 4140036e
|
||||
i.asVecExtract(operandNR(x1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement16B, 8)
|
||||
}},
|
||||
{want: "4138034e", setup: func(i *instruction) {
|
||||
i.asVecPermute(vecOpZip1, operandNR(x1VReg), operandNR(v2VReg), operandNR(v3VReg), vecArrangement16B)
|
||||
}},
|
||||
{want: "4104214f", setup: func(i *instruction) {
|
||||
i.asVecShiftImm(vecOpSshr, operandNR(x1VReg), operandNR(x2VReg), operandShiftImm(31), vecArrangement4S)
|
||||
}},
|
||||
{want: "5b28030b", setup: func(i *instruction) {
|
||||
i.asALU(aluOpAdd, operandNR(tmpRegVReg), operandNR(x2VReg), operandSR(x3VReg, 10, shiftOpLSL), false)
|
||||
}},
|
||||
|
||||
@@ -332,6 +332,21 @@ func (m *machine) LowerInstr(instr *ssa.Instruction) {
|
||||
ins := m.allocateInstr()
|
||||
ins.asVecRRR(vecOpBsl, operandNR(rd), rn, rm, vecArrangement16B)
|
||||
m.insert(ins)
|
||||
case ssa.OpcodeVanyTrue, ssa.OpcodeVallTrue:
|
||||
x, lane := instr.ArgWithLane()
|
||||
var arr vecArrangement
|
||||
if op == ssa.OpcodeVallTrue {
|
||||
arr = ssaLaneToArrangement(lane)
|
||||
}
|
||||
rm := m.getOperand_NR(m.compiler.ValueDefinition(x), extModeNone)
|
||||
rd := operandNR(m.compiler.VRegOf(instr.Return()))
|
||||
m.lowerVcheckTrue(op, rm, rd, arr)
|
||||
case ssa.OpcodeVhighBits:
|
||||
x, lane := instr.ArgWithLane()
|
||||
rm := m.getOperand_NR(m.compiler.ValueDefinition(x), extModeNone)
|
||||
rd := operandNR(m.compiler.VRegOf(instr.Return()))
|
||||
arr := ssaLaneToArrangement(lane)
|
||||
m.lowerVhighBits(rm, rd, arr)
|
||||
case ssa.OpcodeVIadd:
|
||||
x, y, lane := instr.Arg2WithLane()
|
||||
arr := ssaLaneToArrangement(lane)
|
||||
@@ -395,6 +410,262 @@ func (m *machine) LowerInstr(instr *ssa.Instruction) {
|
||||
m.FlushPendingInstructions()
|
||||
}
|
||||
|
||||
func (m *machine) lowerVcheckTrue(op ssa.Opcode, rm, rd operand, arr vecArrangement) {
|
||||
tmp := operandNR(m.compiler.AllocateVReg(regalloc.RegTypeOf(ssa.TypeV128)))
|
||||
|
||||
// Special case VallTrue for i64x2.
|
||||
if op == ssa.OpcodeVallTrue && arr == vecArrangement2D {
|
||||
// cmeq v3?.2d, v2?.2d, #0
|
||||
// addp v3?.2d, v3?.2d, v3?.2d
|
||||
// fcmp x3?, x3?
|
||||
// cset x3?, eq
|
||||
|
||||
ins := m.allocateInstr()
|
||||
ins.asVecMisc(vecOpCmeq0, rd, rm, vecArrangement2D)
|
||||
m.insert(ins)
|
||||
|
||||
addp := m.allocateInstr()
|
||||
addp.asVecRRR(vecOpAddp, rd, rd, rd, vecArrangement2D)
|
||||
m.insert(addp)
|
||||
|
||||
fcmp := m.allocateInstr()
|
||||
fcmp.asFpuCmp(rd, rd, true)
|
||||
m.insert(fcmp)
|
||||
|
||||
cset := m.allocateInstr()
|
||||
cset.asCSet(rd.nr(), eq)
|
||||
m.insert(cset)
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
// Create a scalar value with umaxp or uminv, then compare it against zero.
|
||||
ins := m.allocateInstr()
|
||||
if op == ssa.OpcodeVanyTrue {
|
||||
// umaxp v4?.16b, v2?.16b, v2?.16b
|
||||
ins.asVecRRR(vecOpUmaxp, tmp, rm, rm, vecArrangement16B)
|
||||
} else {
|
||||
// uminv d4?, v2?.4s
|
||||
ins.asVecLanes(vecOpUminv, tmp, rm, arr)
|
||||
}
|
||||
m.insert(ins)
|
||||
|
||||
// mov x3?, v4?.d[0]
|
||||
// ccmp x3?, #0x0, #0x0, al
|
||||
// cset x3?, ne
|
||||
// mov x0, x3?
|
||||
|
||||
movv := m.allocateInstr()
|
||||
movv.asMovFromVec(rd, tmp, vecArrangementD, vecIndex(0))
|
||||
m.insert(movv)
|
||||
|
||||
fc := m.allocateInstr()
|
||||
fc.asCCmpImm(rd, uint64(0), al, 0, true)
|
||||
m.insert(fc)
|
||||
|
||||
cset := m.allocateInstr()
|
||||
cset.asCSet(rd.nr(), ne)
|
||||
m.insert(cset)
|
||||
}
|
||||
|
||||
func (m *machine) lowerVhighBits(rm, rd operand, arr vecArrangement) {
|
||||
r0 := operandNR(m.compiler.AllocateVReg(regalloc.RegTypeOf(ssa.TypeI64)))
|
||||
v0 := operandNR(m.compiler.AllocateVReg(regalloc.RegTypeOf(ssa.TypeV128)))
|
||||
v1 := operandNR(m.compiler.AllocateVReg(regalloc.RegTypeOf(ssa.TypeV128)))
|
||||
|
||||
switch arr {
|
||||
case vecArrangement16B: // ssa.VecLaneI8x16
|
||||
// sshr v6?.16b, v2?.16b, #7
|
||||
// movz x4?, #0x201, lsl 0
|
||||
// movk x4?, #0x804, lsl 16
|
||||
// movk x4?, #0x2010, lsl 32
|
||||
// movk x4?, #0x8040, lsl 48
|
||||
// dup v5?.2d, x4?
|
||||
// and v6?.16b, v6?.16b, v5?.16b
|
||||
// ext v5?.16b, v6?.16b, v6?.16b, #8
|
||||
// zip1 v5?.16b, v6?.16b, v5?.16b
|
||||
// addv s5?, v5?.8h
|
||||
// umov s3?, v5?.h[0]
|
||||
|
||||
// Right arithmetic shift on the original vector and store the result into v1. So we have:
|
||||
// v1[i] = 0xff if vi<0, 0 otherwise.
|
||||
sshr := m.allocateInstr()
|
||||
sshr.asVecShiftImm(vecOpSshr, v1, rm, operandShiftImm(7), vecArrangement16B)
|
||||
m.insert(sshr)
|
||||
|
||||
// Load the bit mask into r0.
|
||||
m.insertMOVZ(r0.nr(), 0x0201, 0, true)
|
||||
m.insertMOVK(r0.nr(), 0x0804, 1, true)
|
||||
m.insertMOVK(r0.nr(), 0x2010, 2, true)
|
||||
m.insertMOVK(r0.nr(), 0x8040, 3, true)
|
||||
|
||||
// dup r0 to v0.
|
||||
dup := m.allocateInstr()
|
||||
dup.asVecDup(v0, r0, vecArrangement2D)
|
||||
m.insert(dup)
|
||||
|
||||
// Lane-wise logical AND with the bit mask, meaning that we have
|
||||
// v[i] = (1 << i) if vi<0, 0 otherwise.
|
||||
//
|
||||
// Below, we use the following notation:
|
||||
// wi := (1 << i) if vi<0, 0 otherwise.
|
||||
and := m.allocateInstr()
|
||||
and.asVecRRR(vecOpAnd, v1, v1, v0, vecArrangement16B)
|
||||
m.insert(and)
|
||||
|
||||
// Swap the lower and higher 8 byte elements, and write it into v0, meaning that we have
|
||||
// v0[i] = w(i+8) if i < 8, w(i-8) otherwise.
|
||||
ext := m.allocateInstr()
|
||||
ext.asVecExtract(v0, v1, v1, vecArrangement16B, uint32(8))
|
||||
m.insert(ext)
|
||||
|
||||
// v = [w0, w8, ..., w7, w15]
|
||||
zip1 := m.allocateInstr()
|
||||
zip1.asVecPermute(vecOpZip1, v0, v1, v0, vecArrangement16B)
|
||||
m.insert(zip1)
|
||||
|
||||
// v.h[0] = w0 + ... + w15
|
||||
addv := m.allocateInstr()
|
||||
addv.asVecLanes(vecOpAddv, v0, v0, vecArrangement8H)
|
||||
m.insert(addv)
|
||||
|
||||
// Extract the v.h[0] as the result.
|
||||
movfv := m.allocateInstr()
|
||||
movfv.asMovFromVec(rd, v0, vecArrangementH, vecIndex(0))
|
||||
m.insert(movfv)
|
||||
case vecArrangement8H: // ssa.VecLaneI16x8
|
||||
// sshr v6?.8h, v2?.8h, #15
|
||||
// movz x4?, #0x1, lsl 0
|
||||
// movk x4?, #0x2, lsl 16
|
||||
// movk x4?, #0x4, lsl 32
|
||||
// movk x4?, #0x8, lsl 48
|
||||
// dup v5?.2d, x4?
|
||||
// lsl x4?, x4?, 0x4
|
||||
// ins v5?.d[1], x4?
|
||||
// and v5?.16b, v6?.16b, v5?.16b
|
||||
// addv s5?, v5?.8h
|
||||
// umov s3?, v5?.h[0]
|
||||
|
||||
// Right arithmetic shift on the original vector and store the result into v1. So we have:
|
||||
// v[i] = 0xffff if vi<0, 0 otherwise.
|
||||
sshr := m.allocateInstr()
|
||||
sshr.asVecShiftImm(vecOpSshr, v1, rm, operandShiftImm(15), vecArrangement8H)
|
||||
m.insert(sshr)
|
||||
|
||||
// Load the bit mask into r0.
|
||||
m.lowerConstantI64(r0.nr(), 0x0008000400020001)
|
||||
|
||||
// dup r0 to vector v0.
|
||||
dup := m.allocateInstr()
|
||||
dup.asVecDup(v0, r0, vecArrangement2D)
|
||||
m.insert(dup)
|
||||
|
||||
lsl := m.allocateInstr()
|
||||
lsl.asALUShift(aluOpLsl, r0, r0, operandShiftImm(4), true)
|
||||
m.insert(lsl)
|
||||
|
||||
movv := m.allocateInstr()
|
||||
movv.asMovToVec(v0, r0, vecArrangementD, vecIndex(1))
|
||||
m.insert(movv)
|
||||
|
||||
// Lane-wise logical AND with the bitmask, meaning that we have
|
||||
// v[i] = (1 << i) if vi<0, 0 otherwise for i=0..3
|
||||
// = (1 << (i+4)) if vi<0, 0 otherwise for i=3..7
|
||||
and := m.allocateInstr()
|
||||
and.asVecRRR(vecOpAnd, v0, v1, v0, vecArrangement16B)
|
||||
m.insert(and)
|
||||
|
||||
addv := m.allocateInstr()
|
||||
addv.asVecLanes(vecOpAddv, v0, v0, vecArrangement8H)
|
||||
m.insert(addv)
|
||||
|
||||
movfv := m.allocateInstr()
|
||||
movfv.asMovFromVec(rd, v0, vecArrangementH, vecIndex(0))
|
||||
m.insert(movfv)
|
||||
case vecArrangement4S: // ssa.VecLaneI32x4
|
||||
// sshr v6?.8h, v2?.8h, #15
|
||||
// movz x4?, #0x1, lsl 0
|
||||
// movk x4?, #0x2, lsl 16
|
||||
// movk x4?, #0x4, lsl 32
|
||||
// movk x4?, #0x8, lsl 48
|
||||
// dup v5?.2d, x4?
|
||||
// lsl x4?, x4?, 0x4
|
||||
// ins v5?.d[1], x4?
|
||||
// and v5?.16b, v6?.16b, v5?.16b
|
||||
// addv s5?, v5?.8h
|
||||
// umov s3?, v5?.h[0]
|
||||
|
||||
// Right arithmetic shift on the original vector and store the result into v1. So we have:
|
||||
// v[i] = 0xffffffff if vi<0, 0 otherwise.
|
||||
sshr := m.allocateInstr()
|
||||
sshr.asVecShiftImm(vecOpSshr, v1, rm, operandShiftImm(31), vecArrangement4S)
|
||||
m.insert(sshr)
|
||||
|
||||
// Load the bit mask into r0.
|
||||
m.lowerConstantI64(r0.nr(), 0x0000000200000001)
|
||||
|
||||
// dup r0 to vector v0.
|
||||
dup := m.allocateInstr()
|
||||
dup.asVecDup(v0, r0, vecArrangement2D)
|
||||
m.insert(dup)
|
||||
|
||||
lsl := m.allocateInstr()
|
||||
lsl.asALUShift(aluOpLsl, r0, r0, operandShiftImm(2), true)
|
||||
m.insert(lsl)
|
||||
|
||||
movv := m.allocateInstr()
|
||||
movv.asMovToVec(v0, r0, vecArrangementD, vecIndex(1))
|
||||
m.insert(movv)
|
||||
|
||||
// Lane-wise logical AND with the bitmask, meaning that we have
|
||||
// v[i] = (1 << i) if vi<0, 0 otherwise for i in [0, 1]
|
||||
// = (1 << (i+4)) if vi<0, 0 otherwise for i in [2, 3]
|
||||
and := m.allocateInstr()
|
||||
and.asVecRRR(vecOpAnd, v0, v1, v0, vecArrangement16B)
|
||||
m.insert(and)
|
||||
|
||||
addv := m.allocateInstr()
|
||||
addv.asVecLanes(vecOpAddv, v0, v0, vecArrangement4S)
|
||||
m.insert(addv)
|
||||
|
||||
movfv := m.allocateInstr()
|
||||
movfv.asMovFromVec(rd, v0, vecArrangementS, vecIndex(0))
|
||||
m.insert(movfv)
|
||||
case vecArrangement2D: // ssa.VecLaneI64x2
|
||||
// mov d3?, v2?.d[0]
|
||||
// mov x4?, v2?.d[1]
|
||||
// lsr x4?, x4?, 0x3f
|
||||
// lsr d3?, d3?, 0x3f
|
||||
// add s3?, s3?, w4?, lsl #1
|
||||
|
||||
// Move the lower 64-bit int into result.
|
||||
movv0 := m.allocateInstr()
|
||||
movv0.asMovFromVec(rd, rm, vecArrangementD, vecIndex(0))
|
||||
m.insert(movv0)
|
||||
|
||||
// Move the higher 64-bit int into r0.
|
||||
movv1 := m.allocateInstr()
|
||||
movv1.asMovFromVec(r0, rm, vecArrangementD, vecIndex(1))
|
||||
m.insert(movv1)
|
||||
|
||||
// Move the sign bit into the least significant bit.
|
||||
lsr1 := m.allocateInstr()
|
||||
lsr1.asALUShift(aluOpLsr, r0, r0, operandShiftImm(63), true)
|
||||
m.insert(lsr1)
|
||||
|
||||
lsr2 := m.allocateInstr()
|
||||
lsr2.asALUShift(aluOpLsr, rd, rd, operandShiftImm(63), true)
|
||||
m.insert(lsr2)
|
||||
|
||||
// rd = (r0<<1) | rd
|
||||
lsl := m.allocateInstr()
|
||||
lsl.asALU(aluOpAdd, rd, rd, operandSR(r0.nr(), 1, shiftOpLSL), false)
|
||||
m.insert(lsl)
|
||||
default:
|
||||
panic("Unsupported " + arr.String())
|
||||
}
|
||||
}
|
||||
|
||||
func (m *machine) lowerVecMisc(op vecOp, instr *ssa.Instruction) {
|
||||
x, lane := instr.ArgWithLane()
|
||||
arr := ssaLaneToArrangement(lane)
|
||||
|
||||
@@ -535,3 +535,193 @@ mul x1.4s, x2.4s, x15.4s
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestMachine_lowerVcheckTrue(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
op ssa.Opcode
|
||||
expectedAsm string
|
||||
arrangement vecArrangement
|
||||
expectedBytes string
|
||||
}{
|
||||
{
|
||||
name: "anyTrue",
|
||||
op: ssa.OpcodeVanyTrue,
|
||||
expectedAsm: `
|
||||
umaxp v1?.16b, x1.16b, x1.16b
|
||||
mov x15, v1?.d[0]
|
||||
ccmp x15, #0x0, #0x0, al
|
||||
cset x15, ne
|
||||
`,
|
||||
expectedBytes: "20a4216e0f3c084ee0e940faef079f9a",
|
||||
},
|
||||
{
|
||||
name: "allTrue 2D",
|
||||
op: ssa.OpcodeVallTrue,
|
||||
arrangement: vecArrangement2D,
|
||||
expectedAsm: `
|
||||
cmeq x15.2d, x1.2d, #0
|
||||
addp x15.2d, x15.2d, x15.2d
|
||||
fcmp x15, x15
|
||||
cset x15, eq
|
||||
`,
|
||||
expectedBytes: "2f98e04eefbdef4ee0216f1eef179f9a",
|
||||
},
|
||||
{
|
||||
name: "allTrue 8B",
|
||||
arrangement: vecArrangement8B,
|
||||
op: ssa.OpcodeVallTrue,
|
||||
expectedAsm: `
|
||||
uminv h1?, x1.8b
|
||||
mov x15, v1?.d[0]
|
||||
ccmp x15, #0x0, #0x0, al
|
||||
cset x15, ne
|
||||
`,
|
||||
expectedBytes: "20a8312e0f3c084ee0e940faef079f9a",
|
||||
},
|
||||
{
|
||||
name: "allTrue 16B",
|
||||
arrangement: vecArrangement16B,
|
||||
op: ssa.OpcodeVallTrue,
|
||||
expectedAsm: `
|
||||
uminv h1?, x1.16b
|
||||
mov x15, v1?.d[0]
|
||||
ccmp x15, #0x0, #0x0, al
|
||||
cset x15, ne
|
||||
`,
|
||||
expectedBytes: "20a8316e0f3c084ee0e940faef079f9a",
|
||||
},
|
||||
{
|
||||
name: "allTrue 4H",
|
||||
arrangement: vecArrangement4H,
|
||||
op: ssa.OpcodeVallTrue,
|
||||
expectedAsm: `
|
||||
uminv s1?, x1.4h
|
||||
mov x15, v1?.d[0]
|
||||
ccmp x15, #0x0, #0x0, al
|
||||
cset x15, ne
|
||||
`,
|
||||
expectedBytes: "20a8712e0f3c084ee0e940faef079f9a",
|
||||
},
|
||||
{
|
||||
name: "allTrue 8H",
|
||||
arrangement: vecArrangement8H,
|
||||
op: ssa.OpcodeVallTrue,
|
||||
expectedAsm: `
|
||||
uminv s1?, x1.8h
|
||||
mov x15, v1?.d[0]
|
||||
ccmp x15, #0x0, #0x0, al
|
||||
cset x15, ne
|
||||
`,
|
||||
expectedBytes: "20a8716e0f3c084ee0e940faef079f9a",
|
||||
},
|
||||
{
|
||||
name: "allTrue 4S",
|
||||
arrangement: vecArrangement4S,
|
||||
op: ssa.OpcodeVallTrue,
|
||||
expectedAsm: `
|
||||
uminv d1?, x1.4s
|
||||
mov x15, v1?.d[0]
|
||||
ccmp x15, #0x0, #0x0, al
|
||||
cset x15, ne
|
||||
`,
|
||||
expectedBytes: "20a8b16e0f3c084ee0e940faef079f9a",
|
||||
},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
_, _, m := newSetupWithMockContext()
|
||||
m.lowerVcheckTrue(tc.op, operandNR(x1VReg), operandNR(x15VReg), tc.arrangement)
|
||||
require.Equal(t, tc.expectedAsm, "\n"+formatEmittedInstructionsInCurrentBlock(m)+"\n")
|
||||
|
||||
m.FlushPendingInstructions()
|
||||
m.encode(m.perBlockHead)
|
||||
buf := m.compiler.Buf()
|
||||
require.Equal(t, tc.expectedBytes, hex.EncodeToString(buf))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestMachine_lowerVhighBits(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
expectedAsm string
|
||||
arrangement vecArrangement
|
||||
expectedBytes string
|
||||
}{
|
||||
{
|
||||
name: "16B",
|
||||
arrangement: vecArrangement16B,
|
||||
expectedAsm: `
|
||||
sshr v3?.16b, x1.16b, #7
|
||||
movz x1?, #0x201, lsl 0
|
||||
movk x1?, #0x804, lsl 16
|
||||
movk x1?, #0x2010, lsl 32
|
||||
movk x1?, #0x8040, lsl 48
|
||||
dup v2?.2d, x1?
|
||||
and v3?.16b, v3?.16b, v2?.16b
|
||||
ext v2?.16b, v3?.16b, v3?.16b, #8
|
||||
zip1 v2?.16b, v3?.16b, v2?.16b
|
||||
addv s2?, v2?.8h
|
||||
umov w15, v2?.h[0]
|
||||
`,
|
||||
expectedBytes: "2004094f204080d28000a1f20002c4f20008f0f2000c084e001c204e0040006e0038004e00b8714e0f3c020e",
|
||||
},
|
||||
{
|
||||
name: "8H",
|
||||
arrangement: vecArrangement8H,
|
||||
expectedAsm: `
|
||||
sshr v3?.8h, x1.8h, #15
|
||||
movz x1?, #0x1, lsl 0
|
||||
movk x1?, #0x2, lsl 16
|
||||
movk x1?, #0x4, lsl 32
|
||||
movk x1?, #0x8, lsl 48
|
||||
dup v2?.2d, x1?
|
||||
lsl x1?, x1?, 0x4
|
||||
ins v2?.d[1], x1?
|
||||
and v2?.16b, v3?.16b, v2?.16b
|
||||
addv s2?, v2?.8h
|
||||
umov w15, v2?.h[0]
|
||||
`,
|
||||
expectedBytes: "2004114f200080d24000a0f28000c0f20001e0f2000c084e00ec7cd3001c184e001c204e00b8714e0f3c020e",
|
||||
},
|
||||
{
|
||||
name: "4S",
|
||||
arrangement: vecArrangement4S,
|
||||
expectedAsm: `
|
||||
sshr v3?.4s, x1.4s, #31
|
||||
movz x1?, #0x1, lsl 0
|
||||
movk x1?, #0x2, lsl 32
|
||||
dup v2?.2d, x1?
|
||||
lsl x1?, x1?, 0x2
|
||||
ins v2?.d[1], x1?
|
||||
and v2?.16b, v3?.16b, v2?.16b
|
||||
addv d2?, v2?.4s
|
||||
umov w15, v2?.s[0]
|
||||
`,
|
||||
expectedBytes: "2004214f200080d24000c0f2000c084e00f47ed3001c184e001c204e00b8b14e0f3c040e",
|
||||
},
|
||||
{
|
||||
name: "2D",
|
||||
arrangement: vecArrangement2D,
|
||||
expectedAsm: `
|
||||
mov x15, x1.d[0]
|
||||
mov x1?, x1.d[1]
|
||||
lsr x1?, x1?, 0x3f
|
||||
lsr x15, x15, 0x3f
|
||||
add w15, w15, w1?, lsl #1
|
||||
`,
|
||||
expectedBytes: "2f3c084e203c184e00fc7fd3effd7fd3ef05000b",
|
||||
},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
_, _, m := newSetupWithMockContext()
|
||||
m.lowerVhighBits(operandNR(x1VReg), operandNR(x15VReg), tc.arrangement)
|
||||
require.Equal(t, tc.expectedAsm, "\n"+formatEmittedInstructionsInCurrentBlock(m)+"\n")
|
||||
|
||||
m.FlushPendingInstructions()
|
||||
m.encode(m.perBlockHead)
|
||||
buf := m.compiler.Buf()
|
||||
require.Equal(t, tc.expectedBytes, hex.EncodeToString(buf))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -141,6 +141,7 @@ func TestSpectestV2(t *testing.T) {
|
||||
{"conversions"},
|
||||
{"if"},
|
||||
{"loop"},
|
||||
{"simd_boolean"},
|
||||
{"simd_bitwise"},
|
||||
{"simd_const"},
|
||||
{"simd_i8x16_arith"},
|
||||
|
||||
@@ -1414,6 +1414,49 @@ func (c *Compiler) lowerCurrentOpcode() {
|
||||
v1 := state.pop()
|
||||
ret := builder.AllocateInstruction().AsVbitselect(c, v1, v2).Insert(builder).Return()
|
||||
state.push(ret)
|
||||
case wasm.OpcodeVecV128AnyTrue:
|
||||
if state.unreachable {
|
||||
break
|
||||
}
|
||||
v1 := state.pop()
|
||||
ret := builder.AllocateInstruction().AsVanyTrue(v1).Insert(builder).Return()
|
||||
state.push(ret)
|
||||
case wasm.OpcodeVecI8x16AllTrue, wasm.OpcodeVecI16x8AllTrue, wasm.OpcodeVecI32x4AllTrue, wasm.OpcodeVecI64x2AllTrue:
|
||||
if state.unreachable {
|
||||
break
|
||||
}
|
||||
var lane ssa.VecLane
|
||||
switch vecOp {
|
||||
case wasm.OpcodeVecI8x16AllTrue:
|
||||
lane = ssa.VecLaneI8x16
|
||||
case wasm.OpcodeVecI16x8AllTrue:
|
||||
lane = ssa.VecLaneI16x8
|
||||
case wasm.OpcodeVecI32x4AllTrue:
|
||||
lane = ssa.VecLaneI32x4
|
||||
case wasm.OpcodeVecI64x2AllTrue:
|
||||
lane = ssa.VecLaneI64x2
|
||||
}
|
||||
v1 := state.pop()
|
||||
ret := builder.AllocateInstruction().AsVallTrue(v1, lane).Insert(builder).Return()
|
||||
state.push(ret)
|
||||
case wasm.OpcodeVecI8x16BitMask, wasm.OpcodeVecI16x8BitMask, wasm.OpcodeVecI32x4BitMask, wasm.OpcodeVecI64x2BitMask:
|
||||
if state.unreachable {
|
||||
break
|
||||
}
|
||||
var lane ssa.VecLane
|
||||
switch vecOp {
|
||||
case wasm.OpcodeVecI8x16BitMask:
|
||||
lane = ssa.VecLaneI8x16
|
||||
case wasm.OpcodeVecI16x8BitMask:
|
||||
lane = ssa.VecLaneI16x8
|
||||
case wasm.OpcodeVecI32x4BitMask:
|
||||
lane = ssa.VecLaneI32x4
|
||||
case wasm.OpcodeVecI64x2BitMask:
|
||||
lane = ssa.VecLaneI64x2
|
||||
}
|
||||
v1 := state.pop()
|
||||
ret := builder.AllocateInstruction().AsVhighBits(v1, lane).Insert(builder).Return()
|
||||
state.push(ret)
|
||||
case wasm.OpcodeVecI8x16Abs, wasm.OpcodeVecI16x8Abs, wasm.OpcodeVecI32x4Abs, wasm.OpcodeVecI64x2Abs:
|
||||
if state.unreachable {
|
||||
break
|
||||
|
||||
@@ -913,8 +913,8 @@ var instructionReturnTypes = [opcodeEnd]returnTypesFn{
|
||||
OpcodeVbnot: returnTypesFnV128,
|
||||
OpcodeVbandnot: returnTypesFnV128,
|
||||
OpcodeVbitselect: returnTypesFnV128,
|
||||
OpcodeVanyTrue: returnTypesFnV128,
|
||||
OpcodeVallTrue: returnTypesFnV128,
|
||||
OpcodeVanyTrue: returnTypesFnI32,
|
||||
OpcodeVallTrue: returnTypesFnI32,
|
||||
OpcodeVhighBits: returnTypesFnV128,
|
||||
OpcodeVIadd: returnTypesFnV128,
|
||||
OpcodeVSaddSat: returnTypesFnV128,
|
||||
@@ -1556,6 +1556,32 @@ func (i *Instruction) AsVbitselect(c, x, y Value) *Instruction {
|
||||
return i
|
||||
}
|
||||
|
||||
// AsVanyTrue initializes this instruction as an anyTrue vector instruction with OpcodeVanyTrue.
|
||||
func (i *Instruction) AsVanyTrue(x Value) *Instruction {
|
||||
i.opcode = OpcodeVanyTrue
|
||||
i.typ = TypeI32
|
||||
i.v = x
|
||||
return i
|
||||
}
|
||||
|
||||
// AsVallTrue initializes this instruction as an allTrue vector instruction with OpcodeVallTrue.
|
||||
func (i *Instruction) AsVallTrue(x Value, lane VecLane) *Instruction {
|
||||
i.opcode = OpcodeVallTrue
|
||||
i.typ = TypeI32
|
||||
i.v = x
|
||||
i.u1 = uint64(lane)
|
||||
return i
|
||||
}
|
||||
|
||||
// AsVhighBits initializes this instruction as a highBits vector instruction with OpcodeVhighBits.
|
||||
func (i *Instruction) AsVhighBits(x Value, lane VecLane) *Instruction {
|
||||
i.opcode = OpcodeVhighBits
|
||||
i.typ = TypeI32
|
||||
i.v = x
|
||||
i.u1 = uint64(lane)
|
||||
return i
|
||||
}
|
||||
|
||||
// VconstData returns the operands of this vector constant instruction.
|
||||
func (i *Instruction) VconstData() (lo, hi uint64) {
|
||||
return i.u1, i.u2
|
||||
@@ -2041,7 +2067,7 @@ func (i *Instruction) Format(b Builder) string {
|
||||
OpcodeCeil, OpcodeFloor, OpcodeTrunc, OpcodeNearest:
|
||||
instSuffix = " " + i.v.Format(b)
|
||||
case OpcodeVIadd, OpcodeVSaddSat, OpcodeVUaddSat, OpcodeVIsub, OpcodeVSsubSat, OpcodeVUsubSat,
|
||||
OpcodeVImin, OpcodeVUmin, OpcodeVImax, OpcodeVUmax, OpcodeVImul:
|
||||
OpcodeVImin, OpcodeVUmin, OpcodeVImax, OpcodeVUmax, OpcodeVImul, OpcodeVAvgRound:
|
||||
instSuffix = fmt.Sprintf(".%s %s, %s", VecLane(i.u1), i.v.Format(b), i.v2.Format(b))
|
||||
case OpcodeVIabs, OpcodeVIneg, OpcodeVIpopcnt, OpcodeVhighBits, OpcodeVallTrue, OpcodeVanyTrue:
|
||||
instSuffix = fmt.Sprintf(".%s %s", VecLane(i.u1), i.v.Format(b))
|
||||
@@ -2437,6 +2463,8 @@ func (o Opcode) String() (ret string) {
|
||||
return "VSsubSat"
|
||||
case OpcodeVUsubSat:
|
||||
return "VUsubSat"
|
||||
case OpcodeVAvgRound:
|
||||
return "OpcodeVAvgRound"
|
||||
case OpcodeVIsub:
|
||||
return "VIsub"
|
||||
case OpcodeVImin:
|
||||
|
||||
Reference in New Issue
Block a user