authorgravatar for andrew@ziglang.orgAndrew Kelley <andrew@ziglang.org> 2022-03-07 04:00:45-05:00
committergravatar for noreply@github.comGitHub <noreply@github.com> 2022-03-07 04:00:45-05:00
log8c32d989c995f8675f1824fb084245b833b26223
tree81f80b17835931b5fa0a877e9c554225f75ea7a9
parent6547da8f97b94453fb08f582c2c7ce4eb1782a80
parent3c1ebf95567db0c844c2618c5b8971d62c27352f
signaturebadge-question-mark Signed by PGP key 4AEE18F83AFDEB23

Merge pull request #11054 from schmee/mul-add

Implement `@mulAdd` for scalar floats

21 files changed, 541 insertions(+), 114 deletions(-)

lib/std/math/fma.zig+2
......@@ -19,6 +19,8 @@ pub fn fma(comptime T: type, x: T, y: T, z: T) T {
1919 // TODO this is not correct for some targets
2020 c_longdouble => @floatCast(c_longdouble, fma128(x, y, z)),
2121
22 f80 => @floatCast(f80, fma128(x, y, z)),
23
2224 else => @compileError("fma not implemented for " ++ @typeName(T)),
2325 };
2426}
lib/std/special/c.zig+1-19
......@@ -12,7 +12,7 @@ const maxInt = std.math.maxInt;
1212const native_os = builtin.os.tag;
1313const native_arch = builtin.cpu.arch;
1414const native_abi = builtin.abi;
15const long_double_is_f128 = builtin.target.longDoubleIsF128();
15const long_double_is_f128 = builtin.target.longDoubleIs(f128);
1616
1717const is_wasm = switch (native_arch) {
1818 .wasm32, .wasm64 => true,
......@@ -90,10 +90,6 @@ comptime {
9090 @export(fmod, .{ .name = "fmod", .linkage = .Strong });
9191 @export(fmodf, .{ .name = "fmodf", .linkage = .Strong });
9292
93 @export(fma, .{ .name = "fma", .linkage = .Strong });
94 @export(fmaf, .{ .name = "fmaf", .linkage = .Strong });
95 @export(fmal, .{ .name = "fmal", .linkage = .Strong });
96
9793 @export(sincos, .{ .name = "sincos", .linkage = .Strong });
9894 @export(sincosf, .{ .name = "sincosf", .linkage = .Strong });
9995
......@@ -561,20 +557,6 @@ test "fmod, fmodf" {
561557 }
562558}
563559
564fn fmaf(a: f32, b: f32, c: f32) callconv(.C) f32 {
565 return math.fma(f32, a, b, c);
566}
567
568fn fma(a: f64, b: f64, c: f64) callconv(.C) f64 {
569 return math.fma(f64, a, b, c);
570}
571fn fmal(a: c_longdouble, b: c_longdouble, c: c_longdouble) callconv(.C) c_longdouble {
572 if (!long_double_is_f128) {
573 @panic("TODO implement this");
574 }
575 return math.fma(c_longdouble, a, b, c);
576}
577
578560fn sincos(a: f64, r_sin: *f64, r_cos: *f64) callconv(.C) void {
579561 r_sin.* = math.sin(a);
580562 r_cos.* = math.cos(a);
lib/std/special/compiler_rt.zig+41-10
......@@ -19,7 +19,8 @@ const strong_linkage = if (is_test)
1919else
2020 std.builtin.GlobalLinkage.Strong;
2121
22const long_double_is_f128 = builtin.target.longDoubleIsF128();
22const long_double_is_f80 = builtin.target.longDoubleIs(f80);
23const long_double_is_f128 = builtin.target.longDoubleIs(f128);
2324
2425comptime {
2526 // These files do their own comptime exporting logic.
......@@ -673,6 +674,29 @@ comptime {
673674 @export(_aullrem, .{ .name = "\x01__aullrem", .linkage = strong_linkage });
674675 }
675676
677 const fmodl = @import("compiler_rt/floatfmodl.zig").fmodl;
678 if (!is_test) {
679 @export(fmodl, .{ .name = "fmodl", .linkage = linkage });
680
681 @export(floorf, .{ .name = "floorf", .linkage = linkage });
682 @export(floor, .{ .name = "floor", .linkage = linkage });
683 @export(floorl, .{ .name = "floorl", .linkage = linkage });
684
685 @export(fma, .{ .name = "fma", .linkage = linkage });
686 @export(fmaf, .{ .name = "fmaf", .linkage = linkage });
687 @export(fmal, .{ .name = "fmal", .linkage = linkage });
688 if (long_double_is_f80) {
689 @export(fmal, .{ .name = "__fmax", .linkage = linkage });
690 } else {
691 @export(__fmax, .{ .name = "__fmax", .linkage = linkage });
692 }
693 if (long_double_is_f128) {
694 @export(fmal, .{ .name = "fmaq", .linkage = linkage });
695 } else {
696 @export(fmaq, .{ .name = "fmaq", .linkage = linkage });
697 }
698 }
699
676700 if (arch.isSPARC()) {
677701 // SPARC systems use a different naming scheme
678702 const _Qp_add = @import("compiler_rt/sparc.zig")._Qp_add;
......@@ -725,7 +749,7 @@ comptime {
725749 @export(_Qp_qtod, .{ .name = "_Qp_qtod", .linkage = linkage });
726750 }
727751
728 if ((arch == .powerpc or arch.isPPC64()) and !is_test) {
752 if ((arch.isPPC() or arch.isPPC64()) and !is_test) {
729753 @export(__addtf3, .{ .name = "__addkf3", .linkage = linkage });
730754 @export(__subtf3, .{ .name = "__subkf3", .linkage = linkage });
731755 @export(__multf3, .{ .name = "__mulkf3", .linkage = linkage });
......@@ -750,22 +774,29 @@ comptime {
750774 @export(__letf2, .{ .name = "__lekf2", .linkage = linkage });
751775 @export(__getf2, .{ .name = "__gtkf2", .linkage = linkage });
752776 @export(__unordtf2, .{ .name = "__unordkf2", .linkage = linkage });
753 }
754777
755 const fmodl = @import("compiler_rt/floatfmodl.zig").fmodl;
756 @export(fmodl, .{ .name = "fmodl", .linkage = linkage });
757
758 @export(floorf, .{ .name = "floorf", .linkage = linkage });
759 @export(floor, .{ .name = "floor", .linkage = linkage });
760 @export(floorl, .{ .name = "floorl", .linkage = linkage });
761 @export(fmaq, .{ .name = "fmaq", .linkage = linkage });
778 // LLVM PPC backend lowers f128 fma to `fmaf128`.
779 @export(fmal, .{ .name = "fmaf128", .linkage = linkage });
780 }
762781}
763782
764783const math = std.math;
765784
785fn fmaf(a: f32, b: f32, c: f32) callconv(.C) f32 {
786 return math.fma(f32, a, b, c);
787}
788fn fma(a: f64, b: f64, c: f64) callconv(.C) f64 {
789 return math.fma(f64, a, b, c);
790}
791fn __fmax(a: f80, b: f80, c: f80) callconv(.C) f80 {
792 return math.fma(f80, a, b, c);
793}
766794fn fmaq(a: f128, b: f128, c: f128) callconv(.C) f128 {
767795 return math.fma(f128, a, b, c);
768796}
797fn fmal(a: c_longdouble, b: c_longdouble, c: c_longdouble) callconv(.C) c_longdouble {
798 return math.fma(c_longdouble, a, b, c);
799}
769800
770801// TODO add intrinsics for these (and probably the double version too)
771802// and have the math stuff use the intrinsic. same as @mod and @rem
lib/std/target.zig+49-3
......@@ -1714,9 +1714,55 @@ pub const Target = struct {
17141714 };
17151715 }
17161716
1717 pub inline fn longDoubleIsF128(target: Target) bool {
1718 return switch (target.cpu.arch) {
1719 .riscv64, .aarch64, .aarch64_be, .aarch64_32, .s390x, .mips64, .mips64el => true,
1717 pub inline fn longDoubleIs(target: Target, comptime F: type) bool {
1718 if (target.abi == .msvc) {
1719 return F == f64;
1720 }
1721 return switch (F) {
1722 f128 => switch (target.cpu.arch) {
1723 .riscv64,
1724 .aarch64,
1725 .aarch64_be,
1726 .aarch64_32,
1727 .s390x,
1728 .mips64,
1729 .mips64el,
1730 .sparc,
1731 .sparcv9,
1732 .sparcel,
1733 .powerpc,
1734 .powerpcle,
1735 .powerpc64,
1736 .powerpc64le,
1737 => true,
1738
1739 else => false,
1740 },
1741 f80 => switch (target.cpu.arch) {
1742 .x86_64, .i386 => true,
1743 else => false,
1744 },
1745 f64 => switch (target.cpu.arch) {
1746 .x86_64,
1747 .i386,
1748 .riscv64,
1749 .aarch64,
1750 .aarch64_be,
1751 .aarch64_32,
1752 .s390x,
1753 .mips64,
1754 .mips64el,
1755 .sparc,
1756 .sparcv9,
1757 .sparcel,
1758 .powerpc,
1759 .powerpcle,
1760 .powerpc64,
1761 .powerpc64le,
1762 => false,
1763
1764 else => true,
1765 },
17201766 else => false,
17211767 };
17221768 }
src/Air.zig+7
......@@ -579,6 +579,11 @@ pub const Inst = struct {
579579 /// Uses the `prefetch` field.
580580 prefetch,
581581
582 /// Computes `(a * b) + c`, but only rounds once.
583 /// Uses the `pl_op` field with payload `Bin`.
584 /// The operand is the addend. The mulends are lhs and rhs.
585 mul_add,
586
582587 /// Implements @fieldParentPtr builtin.
583588 /// Uses the `ty_pl` field.
584589 field_parent_ptr,
......@@ -986,6 +991,8 @@ pub fn typeOfIndex(air: Air, inst: Air.Inst.Index) Type {
986991 return ptr_ty.elemType();
987992 },
988993
994 .mul_add => return air.typeOf(datas[inst].pl_op.operand),
995
989996 .add_with_overflow,
990997 .sub_with_overflow,
991998 .mul_with_overflow,
src/AstGen.zig+2-2
......@@ -7309,8 +7309,8 @@ fn builtinCall(
73097309 },
73107310 .mul_add => {
73117311 const float_type = try typeExpr(gz, scope, params[0]);
7312 const mulend1 = try expr(gz, scope, .{ .ty = float_type }, params[1]);
7313 const mulend2 = try expr(gz, scope, .{ .ty = float_type }, params[2]);
7312 const mulend1 = try expr(gz, scope, .{ .coerced_ty = float_type }, params[1]);
7313 const mulend2 = try expr(gz, scope, .{ .coerced_ty = float_type }, params[2]);
73147314 const addend = try expr(gz, scope, .{ .ty = float_type }, params[3]);
73157315 const result = try gz.addPlNode(.mul_add, node, Zir.Inst.MulAdd{
73167316 .mulend1 = mulend1,
src/Liveness.zig+5
......@@ -464,6 +464,11 @@ fn analyzeInst(
464464 const extra = a.air.extraData(Air.Cmpxchg, inst_datas[inst].ty_pl.payload).data;
465465 return trackOperands(a, new_set, inst, main_tomb, .{ extra.ptr, extra.expected_value, extra.new_value });
466466 },
467 .mul_add => {
468 const pl_op = inst_datas[inst].pl_op;
469 const extra = a.air.extraData(Air.Bin, pl_op.payload).data;
470 return trackOperands(a, new_set, inst, main_tomb, .{ extra.lhs, extra.rhs, pl_op.operand });
471 },
467472 .atomic_load => {
468473 const ptr = inst_datas[inst].atomic_load.ptr;
469474 return trackOperands(a, new_set, inst, main_tomb, .{ ptr, .none, .none });
src/Sema.zig+76-1
......@@ -13518,8 +13518,83 @@ fn zirAtomicStore(sema: *Sema, block: *Block, inst: Zir.Inst.Index) CompileError
1351813518
1351913519fn zirMulAdd(sema: *Sema, block: *Block, inst: Zir.Inst.Index) CompileError!Air.Inst.Ref {
1352013520 const inst_data = sema.code.instructions.items(.data)[inst].pl_node;
13521 const extra = sema.code.extraData(Zir.Inst.MulAdd, inst_data.payload_index).data;
1352113522 const src = inst_data.src();
13522 return sema.fail(block, src, "TODO: Sema.zirMulAdd", .{});
13523
13524 const mulend1_src: LazySrcLoc = .{ .node_offset_builtin_call_arg1 = inst_data.src_node };
13525 const mulend2_src: LazySrcLoc = .{ .node_offset_builtin_call_arg2 = inst_data.src_node };
13526 const addend_src: LazySrcLoc = .{ .node_offset_builtin_call_arg3 = inst_data.src_node };
13527
13528 const addend = sema.resolveInst(extra.addend);
13529 const ty = sema.typeOf(addend);
13530 const mulend1 = try sema.coerce(block, ty, sema.resolveInst(extra.mulend1), mulend1_src);
13531 const mulend2 = try sema.coerce(block, ty, sema.resolveInst(extra.mulend2), mulend2_src);
13532
13533 const target = sema.mod.getTarget();
13534
13535 switch (ty.zigTypeTag()) {
13536 .ComptimeFloat, .Float => {
13537 const maybe_mulend1 = try sema.resolveMaybeUndefVal(block, mulend1_src, mulend1);
13538 const maybe_mulend2 = try sema.resolveMaybeUndefVal(block, mulend2_src, mulend2);
13539 const maybe_addend = try sema.resolveMaybeUndefVal(block, addend_src, addend);
13540
13541 const runtime_src = if (maybe_mulend1) |mulend1_val| rs: {
13542 if (maybe_mulend2) |mulend2_val| {
13543 if (mulend2_val.isUndef()) return sema.addConstUndef(ty);
13544
13545 if (maybe_addend) |addend_val| {
13546 if (addend_val.isUndef()) return sema.addConstUndef(ty);
13547
13548 const result_val = try Value.mulAdd(
13549 ty,
13550 mulend1_val,
13551 mulend2_val,
13552 addend_val,
13553 sema.arena,
13554 target,
13555 );
13556 return sema.addConstant(ty, result_val);
13557 } else {
13558 break :rs addend_src;
13559 }
13560 } else {
13561 if (maybe_addend) |addend_val| {
13562 if (addend_val.isUndef()) return sema.addConstUndef(ty);
13563 }
13564 break :rs mulend2_src;
13565 }
13566 } else rs: {
13567 if (maybe_mulend2) |mulend2_val| {
13568 if (mulend2_val.isUndef()) return sema.addConstUndef(ty);
13569 }
13570 if (maybe_addend) |addend_val| {
13571 if (addend_val.isUndef()) return sema.addConstUndef(ty);
13572 }
13573 break :rs mulend1_src;
13574 };
13575
13576 try sema.requireRuntimeBlock(block, runtime_src);
13577 return block.addInst(.{
13578 .tag = .mul_add,
13579 .data = .{ .pl_op = .{
13580 .operand = addend,
13581 .payload = try sema.addExtra(Air.Bin{
13582 .lhs = mulend1,
13583 .rhs = mulend2,
13584 }),
13585 } },
13586 });
13587 },
13588 .Vector => {
13589 const scalar_ty = ty.scalarType();
13590 switch (scalar_ty.zigTypeTag()) {
13591 .ComptimeFloat, .Float => {},
13592 else => return sema.fail(block, src, "expected vector of floats or float type, found '{}'", .{scalar_ty}),
13593 }
13594 return sema.fail(block, src, "TODO: implement @mulAdd for vectors", .{});
13595 },
13596 else => return sema.fail(block, src, "expected vector of floats or float type, found '{}'", .{ty}),
13597 }
1352313598}
1352413599
1352513600fn zirBuiltinCall(sema: *Sema, block: *Block, inst: Zir.Inst.Index) CompileError!Air.Inst.Ref {
src/Zir.zig+2
......@@ -891,6 +891,8 @@ pub const Inst = struct {
891891 atomic_store,
892892 /// Implements the `@mulAdd` builtin.
893893 /// Uses the `pl_node` union field with payload `MulAdd`.
894 /// The addend communicates the type of the builtin.
895 /// The mulends need to be coerced to the same type.
894896 mul_add,
895897 /// Implements the `@call` builtin.
896898 /// Uses the `pl_node` union field with payload `BuiltinCall`.
src/arch/aarch64/CodeGen.zig+10
......@@ -632,6 +632,7 @@ fn genBody(self: *Self, body: []const Air.Inst.Index) InnerError!void {
632632 .aggregate_init => try self.airAggregateInit(inst),
633633 .union_init => try self.airUnionInit(inst),
634634 .prefetch => try self.airPrefetch(inst),
635 .mul_add => try self.airMulAdd(inst),
635636
636637 .atomic_store_unordered => try self.airAtomicStore(inst, .Unordered),
637638 .atomic_store_monotonic => try self.airAtomicStore(inst, .Monotonic),
......@@ -3652,6 +3653,15 @@ fn airPrefetch(self: *Self, inst: Air.Inst.Index) !void {
36523653 return self.finishAir(inst, MCValue.dead, .{ prefetch.ptr, .none, .none });
36533654}
36543655
3656fn airMulAdd(self: *Self, inst: Air.Inst.Index) !void {
3657 const pl_op = self.air.instructions.items(.data)[inst].pl_op;
3658 const extra = self.air.extraData(Air.Bin, pl_op.payload).data;
3659 const result: MCValue = if (self.liveness.isUnused(inst)) .dead else {
3660 return self.fail("TODO implement airMulAdd for aarch64", .{});
3661 };
3662 return self.finishAir(inst, result, .{ extra.lhs, extra.rhs, pl_op.operand });
3663}
3664
36553665fn resolveInst(self: *Self, inst: Air.Inst.Ref) InnerError!MCValue {
36563666 // First section of indexes correspond to a set number of constant values.
36573667 const ref_int = @enumToInt(inst);
src/arch/arm/CodeGen.zig+10
......@@ -628,6 +628,7 @@ fn genBody(self: *Self, body: []const Air.Inst.Index) InnerError!void {
628628 .aggregate_init => try self.airAggregateInit(inst),
629629 .union_init => try self.airUnionInit(inst),
630630 .prefetch => try self.airPrefetch(inst),
631 .mul_add => try self.airMulAdd(inst),
631632
632633 .atomic_store_unordered => try self.airAtomicStore(inst, .Unordered),
633634 .atomic_store_monotonic => try self.airAtomicStore(inst, .Monotonic),
......@@ -4086,6 +4087,15 @@ fn airPrefetch(self: *Self, inst: Air.Inst.Index) !void {
40864087 return self.finishAir(inst, MCValue.dead, .{ prefetch.ptr, .none, .none });
40874088}
40884089
4090fn airMulAdd(self: *Self, inst: Air.Inst.Index) !void {
4091 const pl_op = self.air.instructions.items(.data)[inst].pl_op;
4092 const extra = self.air.extraData(Air.Bin, pl_op.payload).data;
4093 const result: MCValue = if (self.liveness.isUnused(inst)) .dead else {
4094 return self.fail("TODO implement airMulAdd for arm", .{});
4095 };
4096 return self.finishAir(inst, result, .{ extra.lhs, extra.rhs, pl_op.operand });
4097}
4098
40894099fn resolveInst(self: *Self, inst: Air.Inst.Ref) InnerError!MCValue {
40904100 // First section of indexes correspond to a set number of constant values.
40914101 const ref_int = @enumToInt(inst);
src/arch/riscv64/CodeGen.zig+10
......@@ -600,6 +600,7 @@ fn genBody(self: *Self, body: []const Air.Inst.Index) InnerError!void {
600600 .aggregate_init => try self.airAggregateInit(inst),
601601 .union_init => try self.airUnionInit(inst),
602602 .prefetch => try self.airPrefetch(inst),
603 .mul_add => try self.airMulAdd(inst),
603604
604605 .atomic_store_unordered => try self.airAtomicStore(inst, .Unordered),
605606 .atomic_store_monotonic => try self.airAtomicStore(inst, .Monotonic),
......@@ -2203,6 +2204,15 @@ fn airPrefetch(self: *Self, inst: Air.Inst.Index) !void {
22032204 return self.finishAir(inst, MCValue.dead, .{ prefetch.ptr, .none, .none });
22042205}
22052206
2207fn airMulAdd(self: *Self, inst: Air.Inst.Index) !void {
2208 const pl_op = self.air.instructions.items(.data)[inst].pl_op;
2209 const extra = self.air.extraData(Air.Bin, pl_op.payload).data;
2210 const result: MCValue = if (self.liveness.isUnused(inst)) .dead else {
2211 return self.fail("TODO implement airMulAdd for riscv64", .{});
2212 };
2213 return self.finishAir(inst, result, .{ extra.lhs, extra.rhs, pl_op.operand });
2214}
2215
22062216fn resolveInst(self: *Self, inst: Air.Inst.Ref) InnerError!MCValue {
22072217 // First section of indexes correspond to a set number of constant values.
22082218 const ref_int = @enumToInt(inst);
src/arch/wasm/CodeGen.zig+1
......@@ -1333,6 +1333,7 @@ fn genInst(self: *Self, inst: Air.Inst.Index) !WValue {
13331333 .error_name,
13341334 .errunion_payload_ptr_set,
13351335 .field_parent_ptr,
1336 .mul_add,
13361337
13371338 // For these 4, probably best to wait until https://github.com/ziglang/zig/issues/10248
13381339 // is implemented in the frontend before implementing them here in the wasm backend.
src/arch/x86_64/CodeGen.zig+10
......@@ -717,6 +717,7 @@ fn genBody(self: *Self, body: []const Air.Inst.Index) InnerError!void {
717717 .aggregate_init => try self.airAggregateInit(inst),
718718 .union_init => try self.airUnionInit(inst),
719719 .prefetch => try self.airPrefetch(inst),
720 .mul_add => try self.airMulAdd(inst),
720721
721722 .atomic_store_unordered => try self.airAtomicStore(inst, .Unordered),
722723 .atomic_store_monotonic => try self.airAtomicStore(inst, .Monotonic),
......@@ -5559,6 +5560,15 @@ fn airPrefetch(self: *Self, inst: Air.Inst.Index) !void {
55595560 return self.finishAir(inst, MCValue.dead, .{ prefetch.ptr, .none, .none });
55605561}
55615562
5563fn airMulAdd(self: *Self, inst: Air.Inst.Index) !void {
5564 const pl_op = self.air.instructions.items(.data)[inst].pl_op;
5565 const extra = self.air.extraData(Air.Bin, pl_op.payload).data;
5566 const result: MCValue = if (self.liveness.isUnused(inst)) .dead else {
5567 return self.fail("TODO implement airMulAdd for x86_64", .{});
5568 };
5569 return self.finishAir(inst, result, .{ extra.lhs, extra.rhs, pl_op.operand });
5570}
5571
55625572fn resolveInst(self: *Self, inst: Air.Inst.Ref) InnerError!MCValue {
55635573 // First section of indexes correspond to a set number of constant values.
55645574 const ref_int = @enumToInt(inst);
src/codegen/c.zig+32
......@@ -16,6 +16,7 @@ const trace = @import("../tracy.zig").trace;
1616const LazySrcLoc = Module.LazySrcLoc;
1717const Air = @import("../Air.zig");
1818const Liveness = @import("../Liveness.zig");
19const CType = @import("../type.zig").CType;
1920
2021const Mutability = enum { Const, Mut };
2122const BigIntConst = std.math.big.int.Const;
......@@ -1635,6 +1636,8 @@ fn genBody(f: *Function, body: []const Air.Inst.Index) error{ AnalysisFail, OutO
16351636 .trunc_float,
16361637 => |tag| return f.fail("TODO: C backend: implement unary op for tag '{s}'", .{@tagName(tag)}),
16371638
1639 .mul_add => try airMulAdd(f, inst),
1640
16381641 .add_with_overflow => try airAddWithOverflow(f, inst),
16391642 .sub_with_overflow => try airSubWithOverflow(f, inst),
16401643 .mul_with_overflow => try airMulWithOverflow(f, inst),
......@@ -3621,6 +3624,35 @@ fn airWasmMemoryGrow(f: *Function, inst: Air.Inst.Index) !CValue {
36213624 return local;
36223625}
36233626
3627fn airMulAdd(f: *Function, inst: Air.Inst.Index) !CValue {
3628 if (f.liveness.isUnused(inst)) return CValue.none;
3629 const pl_op = f.air.instructions.items(.data)[inst].pl_op;
3630 const extra = f.air.extraData(Air.Bin, pl_op.payload).data;
3631 const inst_ty = f.air.typeOfIndex(inst);
3632 const mulend1 = try f.resolveInst(extra.lhs);
3633 const mulend2 = try f.resolveInst(extra.rhs);
3634 const addend = try f.resolveInst(pl_op.operand);
3635 const writer = f.object.writer();
3636 const target = f.object.dg.module.getTarget();
3637 const fn_name = switch (inst_ty.floatBits(target)) {
3638 16, 32 => "fmaf",
3639 64 => "fma",
3640 80 => if (CType.longdouble.sizeInBits(target) == 80) "fmal" else "__fmax",
3641 128 => if (CType.longdouble.sizeInBits(target) == 128) "fmal" else "fmaq",
3642 else => unreachable,
3643 };
3644 const local = try f.allocLocal(inst_ty, .Const);
3645 try writer.writeAll(" = ");
3646 try writer.print("{s}(", .{fn_name});
3647 try f.writeCValue(writer, mulend1);
3648 try writer.writeAll(", ");
3649 try f.writeCValue(writer, mulend2);
3650 try writer.writeAll(", ");
3651 try f.writeCValue(writer, addend);
3652 try writer.writeAll(");\n");
3653 return local;
3654}
3655
36243656fn toMemoryOrder(order: std.builtin.AtomicOrder) [:0]const u8 {
36253657 return switch (order) {
36263658 .Unordered => "memory_order_relaxed",
src/codegen/llvm.zig+140-3
......@@ -18,6 +18,7 @@ const target_util = @import("../target.zig");
1818const Value = @import("../value.zig").Value;
1919const Type = @import("../type.zig").Type;
2020const LazySrcLoc = Module.LazySrcLoc;
21const CType = @import("../type.zig").CType;
2122
2223const Error = error{ OutOfMemory, CodegenFail };
2324
......@@ -2189,6 +2190,7 @@ pub const FuncGen = struct {
21892190 .min => try self.airMin(inst),
21902191 .max => try self.airMax(inst),
21912192 .slice => try self.airSlice(inst),
2193 .mul_add => try self.airMulAdd(inst),
21922194
21932195 .add_with_overflow => try self.airOverflow(inst, "llvm.sadd.with.overflow", "llvm.uadd.with.overflow"),
21942196 .sub_with_overflow => try self.airOverflow(inst, "llvm.ssub.with.overflow", "llvm.usub.with.overflow"),
......@@ -3842,6 +3844,46 @@ pub const FuncGen = struct {
38423844 return overflow_bit;
38433845 }
38443846
3847 fn airMulAdd(self: *FuncGen, inst: Air.Inst.Index) !?*const llvm.Value {
3848 if (self.liveness.isUnused(inst)) return null;
3849
3850 const pl_op = self.air.instructions.items(.data)[inst].pl_op;
3851 const extra = self.air.extraData(Air.Bin, pl_op.payload).data;
3852
3853 const mulend1 = try self.resolveInst(extra.lhs);
3854 const mulend2 = try self.resolveInst(extra.rhs);
3855 const addend = try self.resolveInst(pl_op.operand);
3856
3857 const ty = self.air.typeOfIndex(inst);
3858 const llvm_ty = try self.dg.llvmType(ty);
3859 const target = self.dg.module.getTarget();
3860
3861 const Strat = union(enum) {
3862 intrinsic,
3863 libc: [*:0]const u8,
3864 };
3865 const strat: Strat = switch (ty.floatBits(target)) {
3866 16, 32, 64 => Strat.intrinsic,
3867 80 => if (CType.longdouble.sizeInBits(target) == 80) Strat{ .intrinsic = {} } else Strat{ .libc = "__fmax" },
3868 // LLVM always lowers the fma builtin for f128 to fmal, which is for `long double`.
3869 // On some targets this will be correct; on others it will be incorrect.
3870 128 => if (CType.longdouble.sizeInBits(target) == 128) Strat{ .intrinsic = {} } else Strat{ .libc = "fmaq" },
3871 else => unreachable,
3872 };
3873
3874 const llvm_fn = switch (strat) {
3875 .intrinsic => self.getIntrinsic("llvm.fma", &.{llvm_ty}),
3876 .libc => |fn_name| self.dg.object.llvm_module.getNamedFunction(fn_name) orelse b: {
3877 const param_types = [_]*const llvm.Type{ llvm_ty, llvm_ty, llvm_ty };
3878 const fn_type = llvm.functionType(llvm_ty, &param_types, param_types.len, .False);
3879 break :b self.dg.object.llvm_module.addFunction(fn_name, fn_type);
3880 },
3881 };
3882
3883 const params = [_]*const llvm.Value{ mulend1, mulend2, addend };
3884 return self.builder.buildCall(llvm_fn, &params, params.len, .C, .Auto, "");
3885 }
3886
38453887 fn airShlWithOverflow(self: *FuncGen, inst: Air.Inst.Index) !?*const llvm.Value {
38463888 if (self.liveness.isUnused(inst))
38473889 return null;
......@@ -4020,8 +4062,15 @@ pub const FuncGen = struct {
40204062
40214063 const ty_op = self.air.instructions.items(.data)[inst].ty_op;
40224064 const operand = try self.resolveInst(ty_op.operand);
4023 const dest_llvm_ty = try self.dg.llvmType(self.air.typeOfIndex(inst));
4024
4065 const operand_ty = self.air.typeOf(ty_op.operand);
4066 const dest_ty = self.air.typeOfIndex(inst);
4067 const target = self.dg.module.getTarget();
4068 const dest_bits = dest_ty.floatBits(target);
4069 const src_bits = operand_ty.floatBits(target);
4070 if (!backendSupportsF80(target) and (src_bits == 80 or dest_bits == 80)) {
4071 return softF80TruncOrExt(self, operand, src_bits, dest_bits);
4072 }
4073 const dest_llvm_ty = try self.dg.llvmType(dest_ty);
40254074 return self.builder.buildFPTrunc(operand, dest_llvm_ty, "");
40264075 }
40274076
......@@ -4031,8 +4080,15 @@ pub const FuncGen = struct {
40314080
40324081 const ty_op = self.air.instructions.items(.data)[inst].ty_op;
40334082 const operand = try self.resolveInst(ty_op.operand);
4083 const operand_ty = self.air.typeOf(ty_op.operand);
4084 const dest_ty = self.air.typeOfIndex(inst);
4085 const target = self.dg.module.getTarget();
4086 const dest_bits = dest_ty.floatBits(target);
4087 const src_bits = operand_ty.floatBits(target);
4088 if (!backendSupportsF80(target) and (src_bits == 80 or dest_bits == 80)) {
4089 return softF80TruncOrExt(self, operand, src_bits, dest_bits);
4090 }
40344091 const dest_llvm_ty = try self.dg.llvmType(self.air.typeOfIndex(inst));
4035
40364092 return self.builder.buildFPExt(operand, dest_llvm_ty, "");
40374093 }
40384094
......@@ -5064,6 +5120,87 @@ pub const FuncGen = struct {
50645120 return null;
50655121 }
50665122
5123 fn softF80TruncOrExt(
5124 self: *FuncGen,
5125 operand: *const llvm.Value,
5126 src_bits: u16,
5127 dest_bits: u16,
5128 ) !?*const llvm.Value {
5129 const target = self.dg.module.getTarget();
5130
5131 var param_llvm_ty: *const llvm.Type = self.context.intType(80);
5132 var ret_llvm_ty: *const llvm.Type = param_llvm_ty;
5133 var fn_name: [*:0]const u8 = undefined;
5134 var arg = operand;
5135 var final_cast: ?*const llvm.Type = null;
5136
5137 assert(src_bits == 80 or dest_bits == 80);
5138
5139 if (src_bits == 80) switch (dest_bits) {
5140 16 => {
5141 // See corresponding condition at definition of
5142 // __truncxfhf2 in compiler-rt.
5143 if (target.cpu.arch.isAARCH64()) {
5144 ret_llvm_ty = self.context.halfType();
5145 } else {
5146 ret_llvm_ty = self.context.intType(16);
5147 final_cast = self.context.halfType();
5148 }
5149 fn_name = "__truncxfhf2";
5150 },
5151 32 => {
5152 ret_llvm_ty = self.context.floatType();
5153 fn_name = "__truncxfsf2";
5154 },
5155 64 => {
5156 ret_llvm_ty = self.context.doubleType();
5157 fn_name = "__truncxfdf2";
5158 },
5159 80 => return operand,
5160 128 => {
5161 ret_llvm_ty = self.context.fp128Type();
5162 fn_name = "__extendxftf2";
5163 },
5164 else => unreachable,
5165 } else switch (src_bits) {
5166 16 => {
5167 // See corresponding condition at definition of
5168 // __extendhfxf2 in compiler-rt.
5169 param_llvm_ty = if (target.cpu.arch.isAARCH64())
5170 self.context.halfType()
5171 else
5172 self.context.intType(16);
5173 arg = self.builder.buildBitCast(arg, param_llvm_ty, "");
5174 fn_name = "__extendhfxf2";
5175 },
5176 32 => {
5177 param_llvm_ty = self.context.floatType();
5178 fn_name = "__extendsfxf2";
5179 },
5180 64 => {
5181 param_llvm_ty = self.context.doubleType();
5182 fn_name = "__extenddfxf2";
5183 },
5184 80 => return operand,
5185 128 => {
5186 param_llvm_ty = self.context.fp128Type();
5187 fn_name = "__trunctfxf2";
5188 },
5189 else => unreachable,
5190 }
5191
5192 const llvm_fn = self.dg.object.llvm_module.getNamedFunction(fn_name) orelse f: {
5193 const param_types = [_]*const llvm.Type{param_llvm_ty};
5194 const fn_type = llvm.functionType(ret_llvm_ty, &param_types, param_types.len, .False);
5195 break :f self.dg.object.llvm_module.addFunction(fn_name, fn_type);
5196 };
5197
5198 var args: [1]*const llvm.Value = .{arg};
5199 const result = self.builder.buildCall(llvm_fn, &args, args.len, .C, .Auto, "");
5200 const final_cast_llvm_ty = final_cast orelse return result;
5201 return self.builder.buildBitCast(result, final_cast_llvm_ty, "");
5202 }
5203
50675204 fn getErrorNameTable(self: *FuncGen) !*const llvm.Value {
50685205 if (self.dg.object.error_name_table) |table| {
50695206 return table;
src/print_air.zig+12
......@@ -252,6 +252,7 @@ const Writer = struct {
252252 .field_parent_ptr => try w.writeFieldParentPtr(s, inst),
253253 .wasm_memory_size => try w.writeWasmMemorySize(s, inst),
254254 .wasm_memory_grow => try w.writeWasmMemoryGrow(s, inst),
255 .mul_add => try w.writeMulAdd(s, inst),
255256
256257 .add_with_overflow,
257258 .sub_with_overflow,
......@@ -358,6 +359,17 @@ const Writer = struct {
358359 });
359360 }
360361
362 fn writeMulAdd(w: *Writer, s: anytype, inst: Air.Inst.Index) @TypeOf(s).Error!void {
363 const pl_op = w.air.instructions.items(.data)[inst].pl_op;
364 const extra = w.air.extraData(Air.Bin, pl_op.payload).data;
365
366 try w.writeOperand(s, inst, 0, extra.lhs);
367 try s.writeAll(", ");
368 try w.writeOperand(s, inst, 1, extra.rhs);
369 try s.writeAll(", ");
370 try w.writeOperand(s, inst, 2, pl_op.operand);
371 }
372
361373 fn writeFence(w: *Writer, s: anytype, inst: Air.Inst.Index) @TypeOf(s).Error!void {
362374 const atomic_order = w.air.instructions.items(.data)[inst].fence;
363375
src/stage1/target.cpp+10
......@@ -1004,6 +1004,9 @@ bool target_has_debug_info(const ZigTarget *target) {
10041004}
10051005
10061006bool target_long_double_is_f128(const ZigTarget *target) {
1007 if (target->abi == ZigLLVM_MSVC) {
1008 return false;
1009 }
10071010 switch (target->arch) {
10081011 case ZigLLVM_riscv64:
10091012 case ZigLLVM_aarch64:
......@@ -1012,6 +1015,13 @@ bool target_long_double_is_f128(const ZigTarget *target) {
10121015 case ZigLLVM_systemz:
10131016 case ZigLLVM_mips64:
10141017 case ZigLLVM_mips64el:
1018 case ZigLLVM_sparc:
1019 case ZigLLVM_sparcv9:
1020 case ZigLLVM_sparcel:
1021 case ZigLLVM_ppc:
1022 case ZigLLVM_ppcle:
1023 case ZigLLVM_ppc64:
1024 case ZigLLVM_ppc64le:
10151025 return true;
10161026
10171027 default:
src/type.zig+55-67
......@@ -5436,33 +5436,36 @@ pub const CType = enum {
54365436 switch (target.os.tag) {
54375437 .freestanding, .other => switch (target.cpu.arch) {
54385438 .msp430 => switch (self) {
5439 .short,
5440 .ushort,
5441 .int,
5442 .uint,
5443 => return 16,
5444 .long,
5445 .ulong,
5446 => return 32,
5447 .longlong,
5448 .ulonglong,
5449 => return 64,
5450 .longdouble => @panic("TODO figure out what kind of float `long double` is on this target"),
5439 .short, .ushort, .int, .uint => return 16,
5440 .long, .ulong => return 32,
5441 .longlong, .ulonglong, .longdouble => return 64,
54515442 },
54525443 else => switch (self) {
5453 .short,
5454 .ushort,
5455 => return 16,
5456 .int,
5457 .uint,
5458 => return 32,
5459 .long,
5460 .ulong,
5461 => return target.cpu.arch.ptrBitWidth(),
5462 .longlong,
5463 .ulonglong,
5464 => return 64,
5465 .longdouble => @panic("TODO figure out what kind of float `long double` is on this target"),
5444 .short, .ushort => return 16,
5445 .int, .uint => return 32,
5446 .long, .ulong => return target.cpu.arch.ptrBitWidth(),
5447 .longlong, .ulonglong => return 64,
5448 .longdouble => switch (target.cpu.arch) {
5449 .i386, .x86_64 => return 80,
5450
5451 .riscv64,
5452 .aarch64,
5453 .aarch64_be,
5454 .aarch64_32,
5455 .s390x,
5456 .mips64,
5457 .mips64el,
5458 .sparc,
5459 .sparcv9,
5460 .sparcel,
5461 .powerpc,
5462 .powerpcle,
5463 .powerpc64,
5464 .powerpc64le,
5465 => return 128,
5466
5467 else => return 64,
5468 },
54665469 },
54675470 },
54685471
......@@ -5477,19 +5480,13 @@ pub const CType = enum {
54775480 .plan9,
54785481 .solaris,
54795482 => switch (self) {
5480 .short,
5481 .ushort,
5482 => return 16,
5483 .int,
5484 .uint,
5485 => return 32,
5486 .long,
5487 .ulong,
5488 => return target.cpu.arch.ptrBitWidth(),
5489 .longlong,
5490 .ulonglong,
5491 => return 64,
5483 .short, .ushort => return 16,
5484 .int, .uint => return 32,
5485 .long, .ulong => return target.cpu.arch.ptrBitWidth(),
5486 .longlong, .ulonglong => return 64,
54925487 .longdouble => switch (target.cpu.arch) {
5488 .i386, .x86_64 => return 80,
5489
54935490 .riscv64,
54945491 .aarch64,
54955492 .aarch64_be,
......@@ -5497,40 +5494,33 @@ pub const CType = enum {
54975494 .s390x,
54985495 .mips64,
54995496 .mips64el,
5497 .sparc,
5498 .sparcv9,
5499 .sparcel,
5500 .powerpc,
5501 .powerpcle,
5502 .powerpc64,
5503 .powerpc64le,
55005504 => return 128,
55015505
5502 else => return 80,
5506 else => return 64,
55035507 },
55045508 },
55055509
55065510 .windows, .uefi => switch (self) {
5507 .short,
5508 .ushort,
5509 => return 16,
5510 .int,
5511 .uint,
5512 .long,
5513 .ulong,
5514 => return 32,
5515 .longlong,
5516 .ulonglong,
5517 => return 64,
5518 .longdouble => @panic("TODO figure out what kind of float `long double` is on this target"),
5519 },
5520
5521 .ios => switch (self) {
5522 .short,
5523 .ushort,
5524 => return 16,
5525 .int,
5526 .uint,
5527 => return 32,
5528 .long,
5529 .ulong,
5530 .longlong,
5531 .ulonglong,
5532 => return 64,
5533 .longdouble => @panic("TODO figure out what kind of float `long double` is on this target"),
5511 .short, .ushort => return 16,
5512 .int, .uint, .long, .ulong => return 32,
5513 .longlong, .ulonglong, .longdouble => return 64,
5514 },
5515
5516 .ios, .tvos, .watchos => switch (self) {
5517 .short, .ushort => return 16,
5518 .int, .uint => return 32,
5519 .long, .ulong, .longlong, .ulonglong => return 64,
5520 .longdouble => switch (target.cpu.arch) {
5521 .i386, .x86_64 => return 80,
5522 else => return 64,
5523 },
55345524 },
55355525
55365526 .ananas,
......@@ -5549,8 +5539,6 @@ pub const CType = enum {
55495539 .amdhsa,
55505540 .ps4,
55515541 .elfiamcu,
5552 .tvos,
5553 .watchos,
55545542 .mesa3d,
55555543 .contiki,
55565544 .amdpal,
src/value.zig+47-4
......@@ -2931,7 +2931,7 @@ pub const Value = extern union {
29312931 return fromBigInt(arena, result_bigint.toConst());
29322932 }
29332933
2934 /// operands must be integers; handles undefined.
2934 /// operands must be integers; handles undefined.
29352935 pub fn bitwiseAnd(lhs: Value, rhs: Value, arena: Allocator) !Value {
29362936 if (lhs.isUndef() or rhs.isUndef()) return Value.initTag(.undef);
29372937
......@@ -2951,7 +2951,7 @@ pub const Value = extern union {
29512951 return fromBigInt(arena, result_bigint.toConst());
29522952 }
29532953
2954 /// operands must be integers; handles undefined.
2954 /// operands must be integers; handles undefined.
29552955 pub fn bitwiseNand(lhs: Value, rhs: Value, ty: Type, arena: Allocator, target: Target) !Value {
29562956 if (lhs.isUndef() or rhs.isUndef()) return Value.initTag(.undef);
29572957
......@@ -2965,7 +2965,7 @@ pub const Value = extern union {
29652965 return bitwiseXor(anded, all_ones, arena);
29662966 }
29672967
2968 /// operands must be integers; handles undefined.
2968 /// operands must be integers; handles undefined.
29692969 pub fn bitwiseOr(lhs: Value, rhs: Value, arena: Allocator) !Value {
29702970 if (lhs.isUndef() or rhs.isUndef()) return Value.initTag(.undef);
29712971
......@@ -2984,7 +2984,7 @@ pub const Value = extern union {
29842984 return fromBigInt(arena, result_bigint.toConst());
29852985 }
29862986
2987 /// operands must be integers; handles undefined.
2987 /// operands must be integers; handles undefined.
29882988 pub fn bitwiseXor(lhs: Value, rhs: Value, arena: Allocator) !Value {
29892989 if (lhs.isUndef() or rhs.isUndef()) return Value.initTag(.undef);
29902990
......@@ -4020,6 +4020,49 @@ pub const Value = extern union {
40204020 }
40214021 }
40224022
4023 pub fn mulAdd(
4024 float_type: Type,
4025 mulend1: Value,
4026 mulend2: Value,
4027 addend: Value,
4028 arena: Allocator,
4029 target: Target,
4030 ) Allocator.Error!Value {
4031 switch (float_type.floatBits(target)) {
4032 16 => {
4033 const m1 = mulend1.toFloat(f16);
4034 const m2 = mulend2.toFloat(f16);
4035 const a = addend.toFloat(f16);
4036 return Value.Tag.float_16.create(arena, @mulAdd(f16, m1, m2, a));
4037 },
4038 32 => {
4039 const m1 = mulend1.toFloat(f32);
4040 const m2 = mulend2.toFloat(f32);
4041 const a = addend.toFloat(f32);
4042 return Value.Tag.float_32.create(arena, @mulAdd(f32, m1, m2, a));
4043 },
4044 64 => {
4045 const m1 = mulend1.toFloat(f64);
4046 const m2 = mulend2.toFloat(f64);
4047 const a = addend.toFloat(f64);
4048 return Value.Tag.float_64.create(arena, @mulAdd(f64, m1, m2, a));
4049 },
4050 80 => {
4051 const m1 = mulend1.toFloat(f80);
4052 const m2 = mulend2.toFloat(f80);
4053 const a = addend.toFloat(f80);
4054 return Value.Tag.float_80.create(arena, @mulAdd(f80, m1, m2, a));
4055 },
4056 128 => {
4057 const m1 = mulend1.toFloat(f128);
4058 const m2 = mulend2.toFloat(f128);
4059 const a = addend.toFloat(f128);
4060 return Value.Tag.float_128.create(arena, @mulAdd(f128, m1, m2, a));
4061 },
4062 else => unreachable,
4063 }
4064 }
4065
40234066 /// This type is not copyable since it may contain pointers to its inner data.
40244067 pub const Payload = struct {
40254068 tag: Tag,
test/behavior/muladd.zig+19-5
......@@ -2,7 +2,11 @@ const builtin = @import("builtin");
22const expect = @import("std").testing.expect;
33
44test "@mulAdd" {
5 if (builtin.zig_backend != .stage1) return error.SkipZigTest; // TODO
5 if (builtin.zig_backend == .stage2_c) return error.SkipZigTest; // TODO
6 if (builtin.zig_backend == .stage2_wasm) return error.SkipZigTest; // TODO
7 if (builtin.zig_backend == .stage2_x86_64) return error.SkipZigTest; // TODO
8 if (builtin.zig_backend == .stage2_arm) return error.SkipZigTest; // TODO
9 if (builtin.zig_backend == .stage2_aarch64) return error.SkipZigTest; // TODO
610
711 comptime try testMulAdd();
812 try testMulAdd();
......@@ -47,18 +51,28 @@ fn testMulAdd80() !void {
4751}
4852
4953test "@mulAdd f128" {
50 if (builtin.zig_backend != .stage1) return error.SkipZigTest; // TODO
54 if (builtin.zig_backend == .stage2_c) return error.SkipZigTest; // TODO
55 if (builtin.zig_backend == .stage2_wasm) return error.SkipZigTest; // TODO
56 if (builtin.zig_backend == .stage2_x86_64) return error.SkipZigTest; // TODO
57 if (builtin.zig_backend == .stage2_arm) return error.SkipZigTest; // TODO
58 if (builtin.zig_backend == .stage2_aarch64) return error.SkipZigTest; // TODO
5159
5260 if (builtin.os.tag == .macos and builtin.cpu.arch == .aarch64) {
5361 // https://github.com/ziglang/zig/issues/9900
5462 return error.SkipZigTest;
5563 }
5664
57 comptime try testMullAdd128();
58 try testMullAdd128();
65 if (builtin.zig_backend == .stage1 and
66 builtin.cpu.arch == .i386 and builtin.os.tag == .linux)
67 {
68 return error.SkipZigTest;
69 }
70
71 comptime try testMulAdd128();
72 try testMulAdd128();
5973}
6074
61fn testMullAdd128() !void {
75fn testMulAdd128() !void {
6276 var a: f16 = 5.5;
6377 var b: f128 = 2.5;
6478 var c: f128 = 6.25;