Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions cranelift/codegen/src/isa/aarch64/inst.isle
Original file line number Diff line number Diff line change
Expand Up @@ -1715,6 +1715,7 @@
(Abs)
(Neg)
(Sqrt)
(Cvt16To32)
(Cvt32To64)
(Cvt64To32)
))
Expand Down Expand Up @@ -2607,6 +2608,14 @@
(ProducesFlags.ProducesFlagsSideEffect
(MInst.FpuCmp size rn rm)))

;; Half-precision conversions are available even without FEAT_FP16 arithmetic.
(rule 1 (fpu_cmp (ScalarSize.Size16) rn rm)
(if-let false (use_fp16))
(ProducesFlags.ProducesFlagsSideEffect
(MInst.FpuCmp (ScalarSize.Size32)
(fpu_rr (FPUOp1.Cvt16To32) rn (ScalarSize.Size16))
(fpu_rr (FPUOp1.Cvt16To32) rm (ScalarSize.Size16)))))

;; Helper for emitting `MInst.VecLanes` instructions.
(decl vec_lanes (VecLanesOp Reg VectorSize) Reg)
(rule (vec_lanes op src size)
Expand Down
4 changes: 4 additions & 0 deletions cranelift/codegen/src/isa/aarch64/inst/emit.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1853,6 +1853,10 @@ impl MachInstEmit for Inst {
FPUOp1::Abs => 0b000_11110_00_1_000001_10000,
FPUOp1::Neg => 0b000_11110_00_1_000010_10000,
FPUOp1::Sqrt => 0b000_11110_00_1_000011_10000,
FPUOp1::Cvt16To32 => {
debug_assert_eq!(size, ScalarSize::Size16);
0b000_11110_11_1_000100_10000
}
FPUOp1::Cvt32To64 => {
debug_assert_eq!(size, ScalarSize::Size32);
0b000_11110_00_1_000101_10000
Expand Down
11 changes: 11 additions & 0 deletions cranelift/codegen/src/isa/aarch64/inst/emit_tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -6247,6 +6247,17 @@ fn test_aarch64_binemit() {
"fsqrt d15, d30",
));

insns.push((
Inst::FpuRR {
fpu_op: FPUOp1::Cvt16To32,
size: ScalarSize::Size16,
rd: writable_vreg(15),
rn: vreg(30),
},
"CF43E21E",
"fcvt s15, h30",
));

insns.push((
Inst::FpuRR {
fpu_op: FPUOp1::Cvt32To64,
Expand Down
4 changes: 2 additions & 2 deletions cranelift/codegen/src/isa/aarch64/inst/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1787,11 +1787,11 @@ impl Inst {
FPUOp1::Abs => "fabs",
FPUOp1::Neg => "fneg",
FPUOp1::Sqrt => "fsqrt",
FPUOp1::Cvt32To64 | FPUOp1::Cvt64To32 => "fcvt",
FPUOp1::Cvt16To32 | FPUOp1::Cvt32To64 | FPUOp1::Cvt64To32 => "fcvt",
};
let dst_size = match fpu_op {
FPUOp1::Cvt32To64 => ScalarSize::Size64,
FPUOp1::Cvt64To32 => ScalarSize::Size32,
FPUOp1::Cvt16To32 | FPUOp1::Cvt64To32 => ScalarSize::Size32,
_ => size,
};
let rd = pretty_print_vreg_scalar(rd.to_reg(), dst_size);
Expand Down
2 changes: 0 additions & 2 deletions cranelift/codegen/src/isa/aarch64/lower.isle
Original file line number Diff line number Diff line change
Expand Up @@ -1335,7 +1335,6 @@
(rule 1 (lower (uextend (fits_in_64 _) (icmp _ cond x @ (value_type (ty_int _)) y)))
(lower_cond_result_bool (emit_icmp cond x y)))
(rule 2 (lower (uextend (fits_in_64 _) (fcmp _ cond x @ (value_type $F16) y)))
(if-let true (use_fp16))
(lower_cond_result_bool (emit_fcmp cond x y)))
(rule 1 (lower (uextend (fits_in_64 _) (fcmp _ cond x @ (value_type $F32) y)))
(lower_cond_result_bool (emit_fcmp cond x y)))
Expand Down Expand Up @@ -2238,7 +2237,6 @@
(value_reg (float_cmp_zero_swap cond rn vec_size))))

(rule 5 (lower (fcmp _ cond x @ (value_type $F16) y))
(if-let true (use_fp16))
(lower_cond_result_bool (emit_fcmp cond x y)))
(rule 0 (lower (fcmp _ cond x @ (value_type $F32) y))
(lower_cond_result_bool (emit_fcmp cond x y)))
Expand Down
109 changes: 109 additions & 0 deletions cranelift/filetests/filetests/isa/aarch64/fcmp-f16.clif
Original file line number Diff line number Diff line change
@@ -0,0 +1,109 @@
test compile precise-output
set unwind_info=false
target aarch64

function %brif_fcmp_f16(f16, f16) -> i32 {
block0(v0: f16, v1: f16):
v2 = fcmp eq v0, v1
brif v2, block1, block2
block1:
v3 = iconst.i32 1
return v3
block2:
v4 = iconst.i32 0
return v4
}

; VCode:
; block0:
; fcvt s5, h0
; fcvt s7, h1
; fcmp s5, s7
; b.eq label2 ; b label1
; block1:
; movz w0, #0
; ret
; block2:
; movz w0, #1
; ret
;
; Disassembled:
; block0: ; offset 0x0
; fcvt s5, h0
; fcvt s7, h1
; fcmp s5, s7
; b.eq #0x18
; block1: ; offset 0x10
; mov w0, #0
; ret
; block2: ; offset 0x18
; mov w0, #1
; ret

function %select_fcmp_f16(f16, f16, i32, i32) -> i32 {
block0(v0: f16, v1: f16, v2: i32, v3: i32):
v4 = fcmp lt v0, v1
v5 = select v4, v2, v3
return v5
}

; VCode:
; block0:
; fcvt s5, h0
; fcvt s7, h1
; fcmp s5, s7
; csel x0, x0, x1, mi
; ret
;
; Disassembled:
; block0: ; offset 0x0
; fcvt s5, h0
; fcvt s7, h1
; fcmp s5, s7
; csel x0, x0, x1, mi
; ret

function %fcmp_f16(f16, f16) -> i8 {
block0(v0: f16, v1: f16):
v2 = fcmp eq v0, v1
return v2
}

; VCode:
; block0:
; fcvt s3, h0
; fcvt s5, h1
; fcmp s3, s5
; cset x0, eq
; ret
;
; Disassembled:
; block0: ; offset 0x0
; fcvt s3, h0
; fcvt s5, h1
; fcmp s3, s5
; cset x0, eq
; ret

function %uextend_fcmp_f16(f16, f16) -> i32 {
block0(v0: f16, v1: f16):
v2 = fcmp eq v0, v1
v3 = uextend.i32 v2
return v3
}

; VCode:
; block0:
; fcvt s3, h0
; fcvt s5, h1
; fcmp s3, s5
; cset x0, eq
; ret
;
; Disassembled:
; block0: ; offset 0x0
; fcvt s3, h0
; fcvt s5, h1
; fcmp s3, s5
; cset x0, eq
; ret
31 changes: 31 additions & 0 deletions cranelift/filetests/filetests/runtests/fcmp-f16.clif
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
test run
target aarch64
target aarch64 has_fp16

function %brif_fcmp_f16(f16, f16) -> i32 {
block0(v0: f16, v1: f16):
v2 = fcmp eq v0, v1
brif v2, block1, block2
block1:
v3 = iconst.i32 1
return v3
block2:
v4 = iconst.i32 0
return v4
}
; run: %brif_fcmp_f16(0x1.0p0, 0x1.0p0) == 1
; run: %brif_fcmp_f16(0x1.0p0, 0x2.0p0) == 0
; run: %brif_fcmp_f16(0x0.0, -0x0.0) == 1
; run: %brif_fcmp_f16(Inf, Inf) == 1
; run: %brif_fcmp_f16(0x1.0p-24, 0x0.0) == 0
; run: %brif_fcmp_f16(NaN:0x1, 0x1.0p0) == 0

function %select_fcmp_f16(f16, f16, i32, i32) -> i32 {
block0(v0: f16, v1: f16, v2: i32, v3: i32):
v4 = fcmp lt v0, v1
v5 = select v4, v2, v3
return v5
}
; run: %select_fcmp_f16(0x1.0p0, 0x2.0p0, 10, 20) == 10
; run: %select_fcmp_f16(0x2.0p0, 0x1.0p0, 10, 20) == 20
; run: %select_fcmp_f16(NaN:0x1, 0x1.0p0, 10, 20) == 20
Loading