| ... | @@ -895,7 +895,7 @@ fn genFunc(self: *Self) InnerError!void { | ... | @@ -895,7 +895,7 @@ fn genFunc(self: *Self) InnerError!void { |
| 895 | try prologue.append(.{ .tag = .i32_sub, .data = .{ .tag = {} } }); | 895 | try prologue.append(.{ .tag = .i32_sub, .data = .{ .tag = {} } }); |
| 896 | // Get negative stack aligment | 896 | // Get negative stack aligment |
| 897 | try prologue.append(.{ .tag = .i32_const, .data = .{ .imm32 = @intCast(i32, self.stack_alignment) * -1 } }); | 897 | try prologue.append(.{ .tag = .i32_const, .data = .{ .imm32 = @intCast(i32, self.stack_alignment) * -1 } }); |
| 898 | // Bit and the value to get the new stack pointer to ensure the pointers are aligned with the abi alignment | 898 | // Bitwise-and the value to get the new stack pointer to ensure the pointers are aligned with the abi alignment |
| 899 | try prologue.append(.{ .tag = .i32_and, .data = .{ .tag = {} } }); | 899 | try prologue.append(.{ .tag = .i32_and, .data = .{ .tag = {} } }); |
| 900 | // store the current stack pointer as the bottom, which will be used to calculate all stack pointer offsets | 900 | // store the current stack pointer as the bottom, which will be used to calculate all stack pointer offsets |
| 901 | try prologue.append(.{ .tag = .local_tee, .data = .{ .label = self.bottom_stack_value.local } }); | 901 | try prologue.append(.{ .tag = .local_tee, .data = .{ .label = self.bottom_stack_value.local } }); |
| ... | @@ -1074,22 +1074,123 @@ fn toWasmBits(bits: u16) ?u16 { | ... | @@ -1074,22 +1074,123 @@ fn toWasmBits(bits: u16) ?u16 { |
| 1074 | | 1074 | |
| 1075 | /// Performs a copy of bytes for a given type. Copying all bytes | 1075 | /// Performs a copy of bytes for a given type. Copying all bytes |
| 1076 | /// from rhs to lhs. | 1076 | /// from rhs to lhs. |
| 1077 | /// | 1077 | fn memcpy(self: *Self, dst: WValue, src: WValue, len: WValue) !void { |
| 1078 | /// TODO: Perform feature detection and when bulk_memory is available, | 1078 | // When bulk_memory is enabled, we lower it to wasm's memcpy instruction. |
| 1079 | /// use wasm's mem.copy instruction. | 1079 | // If not, we lower it ourselves manually |
| 1080 | fn memCopy(self: *Self, ty: Type, lhs: WValue, rhs: WValue) !void { | 1080 | if (std.Target.wasm.featureSetHas(self.target.cpu.features, .bulk_memory)) { |
| 1081 | const abi_size = ty.abiSize(self.target); | 1081 | switch (dst) { |
| 1082 | var offset: u32 = 0; | 1082 | .stack_offset => try self.emitWValue(try self.buildPointerOffset(dst, 0, .new)), |
| 1083 | const lhs_base = lhs.offset(); | 1083 | else => try self.emitWValue(dst), |
| 1084 | const rhs_base = rhs.offset(); | 1084 | } |
| 1085 | while (offset < abi_size) : (offset += 1) { | 1085 | switch (src) { |
| 1086 | // get lhs' address to store the result | 1086 | .stack_offset => try self.emitWValue(try self.buildPointerOffset(src, 0, .new)), |
| 1087 | try self.emitWValue(lhs); | 1087 | else => try self.emitWValue(src), |
| 1088 | // load byte from rhs' adress | 1088 | } |
| 1089 | try self.emitWValue(rhs); | 1089 | try self.emitWValue(len); |
| 1090 | try self.addMemArg(.i32_load8_u, .{ .offset = rhs_base + offset, .alignment = 1 }); | 1090 | try self.addExtended(.memory_copy); |
| 1091 | // store the result in lhs (we already have its address on the stack) | 1091 | return; |
| 1092 | try self.addMemArg(.i32_store8, .{ .offset = lhs_base + offset, .alignment = 1 }); | 1092 | } |
| | 1093 | |
| | 1094 | // when the length is comptime-known, rather than a runtime value, we can optimize the generated code by having |
| | 1095 | // the loop during codegen, rather than inserting a runtime loop into the binary. |
| | 1096 | switch (len) { |
| | 1097 | .imm32, .imm64 => { |
| | 1098 | const length = switch (len) { |
| | 1099 | .imm32 => |val| val, |
| | 1100 | .imm64 => |val| val, |
| | 1101 | else => unreachable, |
| | 1102 | }; |
| | 1103 | var offset: u32 = 0; |
| | 1104 | const lhs_base = dst.offset(); |
| | 1105 | const rhs_base = src.offset(); |
| | 1106 | while (offset < length) : (offset += 1) { |
| | 1107 | // get dst's address to store the result |
| | 1108 | try self.emitWValue(dst); |
| | 1109 | // load byte from src's address |
| | 1110 | try self.emitWValue(src); |
| | 1111 | switch (self.arch()) { |
| | 1112 | .wasm32 => { |
| | 1113 | try self.addMemArg(.i32_load8_u, .{ .offset = rhs_base + offset, .alignment = 1 }); |
| | 1114 | try self.addMemArg(.i32_store8, .{ .offset = lhs_base + offset, .alignment = 1 }); |
| | 1115 | }, |
| | 1116 | .wasm64 => { |
| | 1117 | try self.addMemArg(.i64_load8_u, .{ .offset = rhs_base + offset, .alignment = 1 }); |
| | 1118 | try self.addMemArg(.i64_store8, .{ .offset = lhs_base + offset, .alignment = 1 }); |
| | 1119 | }, |
| | 1120 | else => unreachable, |
| | 1121 | } |
| | 1122 | } |
| | 1123 | }, |
| | 1124 | else => { |
| | 1125 | // TODO: We should probably lower this to a call to compiler_rt |
| | 1126 | // But for now, we implement it manually |
| | 1127 | const offset = try self.allocLocal(Type.usize); // local for counter |
| | 1128 | // outer block to jump to when loop is done |
| | 1129 | try self.startBlock(.block, wasm.block_empty); |
| | 1130 | try self.startBlock(.loop, wasm.block_empty); |
| | 1131 | |
| | 1132 | // loop condition (offset == length -> break) |
| | 1133 | { |
| | 1134 | try self.emitWValue(offset); |
| | 1135 | try self.emitWValue(len); |
| | 1136 | switch (self.arch()) { |
| | 1137 | .wasm32 => try self.addTag(.i32_eq), |
| | 1138 | .wasm64 => try self.addTag(.i64_eq), |
| | 1139 | else => unreachable, |
| | 1140 | } |
| | 1141 | try self.addLabel(.br_if, 1); // jump out of loop into outer block (finished) |
| | 1142 | } |
| | 1143 | |
| | 1144 | // get dst ptr |
| | 1145 | { |
| | 1146 | try self.emitWValue(dst); |
| | 1147 | try self.emitWValue(offset); |
| | 1148 | switch (self.arch()) { |
| | 1149 | .wasm32 => try self.addTag(.i32_add), |
| | 1150 | .wasm64 => try self.addTag(.i64_add), |
| | 1151 | else => unreachable, |
| | 1152 | } |
| | 1153 | } |
| | 1154 | |
| | 1155 | // get src value and also store in dst |
| | 1156 | { |
| | 1157 | try self.emitWValue(src); |
| | 1158 | try self.emitWValue(offset); |
| | 1159 | switch (self.arch()) { |
| | 1160 | .wasm32 => { |
| | 1161 | try self.addTag(.i32_add); |
| | 1162 | try self.addMemArg(.i32_load8_u, .{ .offset = src.offset(), .alignment = 1 }); |
| | 1163 | try self.addMemArg(.i32_store8, .{ .offset = dst.offset(), .alignment = 1 }); |
| | 1164 | }, |
| | 1165 | .wasm64 => { |
| | 1166 | try self.addTag(.i64_add); |
| | 1167 | try self.addMemArg(.i64_load8_u, .{ .offset = src.offset(), .alignment = 1 }); |
| | 1168 | try self.addMemArg(.i64_store8, .{ .offset = dst.offset(), .alignment = 1 }); |
| | 1169 | }, |
| | 1170 | else => unreachable, |
| | 1171 | } |
| | 1172 | } |
| | 1173 | |
| | 1174 | // increment loop counter |
| | 1175 | { |
| | 1176 | try self.emitWValue(offset); |
| | 1177 | switch (self.arch()) { |
| | 1178 | .wasm32 => { |
| | 1179 | try self.addImm32(1); |
| | 1180 | try self.addTag(.i32_add); |
| | 1181 | }, |
| | 1182 | .wasm64 => { |
| | 1183 | try self.addImm64(1); |
| | 1184 | try self.addTag(.i64_add); |
| | 1185 | }, |
| | 1186 | else => unreachable, |
| | 1187 | } |
| | 1188 | try self.addLabel(.local_set, offset.local); |
| | 1189 | try self.addLabel(.br, 0); // jump to start of loop |
| | 1190 | } |
| | 1191 | try self.endBlock(); // close off loop block |
| | 1192 | try self.endBlock(); // close off outer block |
| | 1193 | }, |
| 1093 | } | 1194 | } |
| 1094 | } | 1195 | } |
| 1095 | | 1196 | |
| ... | @@ -1297,6 +1398,8 @@ fn genInst(self: *Self, inst: Air.Inst.Index) !WValue { | ... | @@ -1297,6 +1398,8 @@ fn genInst(self: *Self, inst: Air.Inst.Index) !WValue { |
| 1297 | .wasm_memory_size => self.airWasmMemorySize(inst), | 1398 | .wasm_memory_size => self.airWasmMemorySize(inst), |
| 1298 | .wasm_memory_grow => self.airWasmMemoryGrow(inst), | 1399 | .wasm_memory_grow => self.airWasmMemoryGrow(inst), |
| 1299 | | 1400 | |
| | 1401 | .memcpy => self.airMemcpy(inst), |
| | 1402 | |
| 1300 | .add_sat, | 1403 | .add_sat, |
| 1301 | .sub_sat, | 1404 | .sub_sat, |
| 1302 | .mul_sat, | 1405 | .mul_sat, |
| ... | @@ -1337,7 +1440,6 @@ fn genInst(self: *Self, inst: Air.Inst.Index) !WValue { | ... | @@ -1337,7 +1440,6 @@ fn genInst(self: *Self, inst: Air.Inst.Index) !WValue { |
| 1337 | .ptr_slice_len_ptr, | 1440 | .ptr_slice_len_ptr, |
| 1338 | .ptr_slice_ptr_ptr, | 1441 | .ptr_slice_ptr_ptr, |
| 1339 | .int_to_float, | 1442 | .int_to_float, |
| 1340 | .memcpy, | | |
| 1341 | .cmpxchg_weak, | 1443 | .cmpxchg_weak, |
| 1342 | .cmpxchg_strong, | 1444 | .cmpxchg_strong, |
| 1343 | .fence, | 1445 | .fence, |
| ... | @@ -1519,7 +1621,8 @@ fn store(self: *Self, lhs: WValue, rhs: WValue, ty: Type, offset: u32) InnerErro | ... | @@ -1519,7 +1621,8 @@ fn store(self: *Self, lhs: WValue, rhs: WValue, ty: Type, offset: u32) InnerErro |
| 1519 | return self.store(lhs, rhs, err_ty, 0); | 1621 | return self.store(lhs, rhs, err_ty, 0); |
| 1520 | } | 1622 | } |
| 1521 | | 1623 | |
| 1522 | return self.memCopy(ty, lhs, rhs); | 1624 | const len = @intCast(u32, ty.abiSize(self.target)); |
| | 1625 | return self.memcpy(lhs, rhs, .{ .imm32 = len }); |
| 1523 | }, | 1626 | }, |
| 1524 | .Optional => { | 1627 | .Optional => { |
| 1525 | if (ty.isPtrLikeOptional()) { | 1628 | if (ty.isPtrLikeOptional()) { |
| ... | @@ -1531,10 +1634,12 @@ fn store(self: *Self, lhs: WValue, rhs: WValue, ty: Type, offset: u32) InnerErro | ... | @@ -1531,10 +1634,12 @@ fn store(self: *Self, lhs: WValue, rhs: WValue, ty: Type, offset: u32) InnerErro |
| 1531 | return self.store(lhs, rhs, Type.u8, 0); | 1634 | return self.store(lhs, rhs, Type.u8, 0); |
| 1532 | } | 1635 | } |
| 1533 | | 1636 | |
| 1534 | return self.memCopy(ty, lhs, rhs); | 1637 | const len = @intCast(u32, ty.abiSize(self.target)); |
| | 1638 | return self.memcpy(lhs, rhs, .{ .imm32 = len }); |
| 1535 | }, | 1639 | }, |
| 1536 | .Struct, .Array, .Union, .Vector => { | 1640 | .Struct, .Array, .Union, .Vector => { |
| 1537 | return self.memCopy(ty, lhs, rhs); | 1641 | const len = @intCast(u32, ty.abiSize(self.target)); |
| | 1642 | return self.memcpy(lhs, rhs, .{ .imm32 = len }); |
| 1538 | }, | 1643 | }, |
| 1539 | .Pointer => { | 1644 | .Pointer => { |
| 1540 | if (ty.isSlice()) { | 1645 | if (ty.isSlice()) { |
| ... | @@ -1549,7 +1654,8 @@ fn store(self: *Self, lhs: WValue, rhs: WValue, ty: Type, offset: u32) InnerErro | ... | @@ -1549,7 +1654,8 @@ fn store(self: *Self, lhs: WValue, rhs: WValue, ty: Type, offset: u32) InnerErro |
| 1549 | } | 1654 | } |
| 1550 | }, | 1655 | }, |
| 1551 | .Int => if (ty.intInfo(self.target).bits > 64) { | 1656 | .Int => if (ty.intInfo(self.target).bits > 64) { |
| 1552 | return self.memCopy(ty, lhs, rhs); | 1657 | const len = @intCast(u32, ty.abiSize(self.target)); |
| | 1658 | return self.memcpy(lhs, rhs, .{ .imm32 = len }); |
| 1553 | }, | 1659 | }, |
| 1554 | else => {}, | 1660 | else => {}, |
| 1555 | } | 1661 | } |
| ... | @@ -3300,3 +3406,13 @@ fn airFieldParentPtr(self: *Self, inst: Air.Inst.Index) InnerError!WValue { | ... | @@ -3300,3 +3406,13 @@ fn airFieldParentPtr(self: *Self, inst: Air.Inst.Index) InnerError!WValue { |
| 3300 | try self.addLabel(.local_set, base.local); | 3406 | try self.addLabel(.local_set, base.local); |
| 3301 | return base; | 3407 | return base; |
| 3302 | } | 3408 | } |
| | 3409 | |
| | 3410 | fn airMemcpy(self: *Self, inst: Air.Inst.Index) InnerError!WValue { |
| | 3411 | const pl_op = self.air.instructions.items(.data)[inst].pl_op; |
| | 3412 | const bin_op = self.air.extraData(Air.Bin, pl_op.payload).data; |
| | 3413 | const dst = try self.resolveInst(pl_op.operand); |
| | 3414 | const src = try self.resolveInst(bin_op.lhs); |
| | 3415 | const len = try self.resolveInst(bin_op.rhs); |
| | 3416 | try self.memcpy(dst, src, len); |
| | 3417 | return WValue{ .none = {} }; |
| | 3418 | } |